test: add integration and E2E test infrastructure
Build and Push API / release (push) Successful in 3s
Build and Push API / build (push) Has been cancelled

- Add pytest markers (integration, e2e, slow) with skip logic
- Add command line options (--run-integration, --run-e2e)
- Create sample_project and sample_project_with_bug fixtures
- Add test_integration.py with 10 LLM tests
- Add test_e2e.py with 12 API server tests
- Update COVERAGE.md to reflect ~65% complete

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
2026-01-11 19:33:21 +01:00
co-authored by Claude Opus 4.5
parent ef69d9c945
commit c839e263f9
5 changed files with 748 additions and 8 deletions
+23 -6
View File
@@ -2,7 +2,7 @@
> Tracking progress towards Claude Code-like functionality > Tracking progress towards Claude Code-like functionality
## Current Status: ~60% Complete ## Current Status: ~65% Complete
Last updated: 2026-01-11 Last updated: 2026-01-11
@@ -75,7 +75,7 @@ Last updated: 2026-01-11
| `GET /agents/{name}` | ✅ | Get agent info | | `GET /agents/{name}` | ✅ | Get agent info |
| Request/response schemas | ✅ | Pydantic models | | Request/response schemas | ✅ | Pydantic models |
### Phase 6: Polish & Tests ⚠️ Partial ### Phase 6: Polish & Tests ✅ Complete
| Component | Status | Notes | | Component | Status | Notes |
|-----------|--------|-------| |-----------|--------|-------|
@@ -83,8 +83,8 @@ Last updated: 2026-01-11
| API endpoint tests | ✅ | 11 tests for agent routes | | API endpoint tests | ✅ | 11 tests for agent routes |
| Health check tests | ✅ | 2 tests | | Health check tests | ✅ | 2 tests |
| Security tests | ✅ | 14 tests for path traversal, injection | | Security tests | ✅ | 14 tests for path traversal, injection |
| Integration tests | | No real LLM integration tests | | Integration tests | | 10 tests with real LLM (requires Ollama) |
| CLI E2E tests | | Not implemented | | E2E tests | | 12 tests against running API server |
--- ---
@@ -130,8 +130,8 @@ Last updated: 2026-01-11
| Tool unit tests | 109 | 109 | ✅ | | Tool unit tests | 109 | 109 | ✅ |
| API tests | 11 | 11 | ✅ | | API tests | 11 | 11 | ✅ |
| Security tests | 14 | 14 | ✅ | | Security tests | 14 | 14 | ✅ |
| Integration tests | 0 | 5 | Agent + real LLM tests | | Integration tests | 10 | 10 | ✅ Agent + real LLM |
| CLI E2E tests | 0 | 10 | Full workflow tests | | E2E tests | 12 | 12 | Full API workflow |
**Test breakdown:** **Test breakdown:**
- Read/Glob/Grep tools: 17 tests - Read/Glob/Grep tools: 17 tests
@@ -142,6 +142,23 @@ Last updated: 2026-01-11
- API endpoints: 11 tests - API endpoints: 11 tests
- Security: 14 tests - Security: 14 tests
- Health checks: 2 tests - Health checks: 2 tests
- Integration (LLM): 10 tests
- E2E (API): 12 tests
**Running tests:**
```bash
# Unit tests only (default)
pytest tests/
# Include integration tests (requires Ollama)
pytest tests/ --run-integration
# Include E2E tests (requires running API server)
pytest tests/ --run-e2e
# All tests
pytest tests/ --run-integration --run-e2e
```
--- ---
+9 -1
View File
@@ -27,7 +27,15 @@ include = ["src*"]
testpaths = ["tests"] testpaths = ["tests"]
python_files = ["test_*.py"] python_files = ["test_*.py"]
python_functions = ["test_*"] python_functions = ["test_*"]
addopts = "-v" addopts = "-v --strict-markers"
markers = [
"integration: marks tests as integration tests (require Ollama to be running)",
"e2e: marks tests as end-to-end tests (require API server to be running)",
"slow: marks tests as slow (may take > 10 seconds)",
]
filterwarnings = [
"ignore::pytest.PytestUnraisableExceptionWarning",
]
[tool.mypy] [tool.mypy]
python_version = "3.12" python_version = "3.12"
+310 -1
View File
@@ -1,12 +1,75 @@
""" """
Pytest configuration and fixtures. Pytest configuration and fixtures.
Test categories:
- Unit tests: Run by default, no external dependencies
- Integration tests: Require Ollama, run with --run-integration
- E2E tests: Require running API server, run with --run-e2e
Usage:
pytest tests/ # Run unit tests only
pytest tests/ --run-integration # Include integration tests
pytest tests/ --run-e2e # Include E2E tests
pytest tests/ --run-integration --run-e2e # Run all tests
""" """
import os
import tempfile
from pathlib import Path
import pytest import pytest
from httpx import ASGITransport, AsyncClient from httpx import ASGITransport, AsyncClient
from src.main import app from src.main import app
# =============================================================================
# Command Line Options
# =============================================================================
def pytest_addoption(parser):
"""Add custom command line options."""
parser.addoption(
"--run-integration",
action="store_true",
default=False,
help="Run integration tests (require Ollama to be running)",
)
parser.addoption(
"--run-e2e",
action="store_true",
default=False,
help="Run E2E tests (require API server to be running)",
)
parser.addoption(
"--ollama-url",
action="store",
default="http://192.168.86.149:11434",
help="Ollama API URL for integration tests",
)
parser.addoption(
"--api-url",
action="store",
default="http://localhost:8095",
help="Webber API URL for E2E tests",
)
def pytest_collection_modifyitems(config, items):
"""Skip integration/e2e tests unless explicitly requested."""
skip_integration = pytest.mark.skip(reason="need --run-integration option to run")
skip_e2e = pytest.mark.skip(reason="need --run-e2e option to run")
for item in items:
if "integration" in item.keywords and not config.getoption("--run-integration"):
item.add_marker(skip_integration)
if "e2e" in item.keywords and not config.getoption("--run-e2e"):
item.add_marker(skip_e2e)
# =============================================================================
# Basic Fixtures
# =============================================================================
@pytest.fixture @pytest.fixture
def anyio_backend(): def anyio_backend():
"""Use asyncio for async tests.""" """Use asyncio for async tests."""
@@ -15,7 +78,7 @@ def anyio_backend():
@pytest.fixture @pytest.fixture
async def client(): async def client():
"""Async HTTP client for testing.""" """Async HTTP client for testing (no auth)."""
async with AsyncClient( async with AsyncClient(
transport=ASGITransport(app=app), transport=ASGITransport(app=app),
base_url="http://test" base_url="http://test"
@@ -32,3 +95,249 @@ async def auth_client():
headers={"X-API-Key": "test-api-key"} headers={"X-API-Key": "test-api-key"}
) as ac: ) as ac:
yield ac yield ac
# =============================================================================
# Integration Test Fixtures
# =============================================================================
@pytest.fixture
def ollama_url(request):
"""Get Ollama URL from command line or environment."""
return request.config.getoption("--ollama-url") or os.environ.get(
"OLLAMA_URL", "http://192.168.86.149:11434"
)
@pytest.fixture
def api_url(request):
"""Get API URL from command line or environment."""
return request.config.getoption("--api-url") or os.environ.get(
"WEBBER_API_URL", "http://localhost:8095"
)
@pytest.fixture
def sample_project():
"""Create a sample Python project for testing."""
with tempfile.TemporaryDirectory() as tmpdir:
project = Path(tmpdir)
# Create a realistic project structure
(project / "src").mkdir()
(project / "tests").mkdir()
# Main application file
(project / "src" / "__init__.py").write_text("")
(project / "src" / "main.py").write_text('''"""Main application module."""
def greet(name: str) -> str:
"""Greet someone by name.
Args:
name: The name to greet
Returns:
A greeting string
"""
return f"Hello, {name}!"
def add(a: int, b: int) -> int:
"""Add two numbers.
Args:
a: First number
b: Second number
Returns:
Sum of a and b
"""
return a + b
def divide(a: float, b: float) -> float:
"""Divide two numbers.
Args:
a: Dividend
b: Divisor
Returns:
Result of a / b
Raises:
ValueError: If b is zero
"""
if b == 0:
raise ValueError("Cannot divide by zero")
return a / b
if __name__ == "__main__":
print(greet("World"))
''')
# Utility module
(project / "src" / "utils.py").write_text('''"""Utility functions."""
def is_even(n: int) -> bool:
"""Check if a number is even."""
return n % 2 == 0
def is_prime(n: int) -> bool:
"""Check if a number is prime."""
if n < 2:
return False
for i in range(2, int(n ** 0.5) + 1):
if n % i == 0:
return False
return True
def factorial(n: int) -> int:
"""Calculate factorial recursively."""
if n <= 1:
return 1
return n * factorial(n - 1)
def fibonacci(n: int) -> list[int]:
"""Generate Fibonacci sequence up to n terms."""
if n <= 0:
return []
if n == 1:
return [0]
fib = [0, 1]
for _ in range(2, n):
fib.append(fib[-1] + fib[-2])
return fib
''')
# Test file
(project / "tests" / "__init__.py").write_text("")
(project / "tests" / "test_main.py").write_text('''"""Tests for main module."""
import pytest
from src.main import greet, add, divide
def test_greet():
assert greet("World") == "Hello, World!"
def test_add():
assert add(2, 3) == 5
def test_divide():
assert divide(10, 2) == 5.0
def test_divide_by_zero():
with pytest.raises(ValueError):
divide(1, 0)
''')
# README
(project / "README.md").write_text('''# Sample Project
A simple Python project for testing Webber's code exploration.
## Features
- Greeting functionality
- Math utilities
- Comprehensive test suite
## Usage
```python
from src.main import greet, add
print(greet("World"))
print(add(2, 3))
```
## Testing
```bash
pytest tests/
```
''')
# Configuration files
(project / "pyproject.toml").write_text('''[project]
name = "sample-project"
version = "0.1.0"
[tool.pytest.ini_options]
testpaths = ["tests"]
''')
yield project
@pytest.fixture
def sample_project_with_bug():
"""Create a sample project with intentional bugs for testing."""
with tempfile.TemporaryDirectory() as tmpdir:
project = Path(tmpdir)
(project / "buggy.py").write_text('''"""Module with intentional bugs."""
def divide_numbers(a, b):
"""Divide two numbers - BUG: no zero check."""
return a / b # BUG: ZeroDivisionError if b is 0
def get_item(lst, index):
"""Get item from list - BUG: no bounds check."""
return lst[index] # BUG: IndexError if out of bounds
def parse_int(s):
"""Parse string to int - BUG: no error handling."""
return int(s) # BUG: ValueError if not a valid int
# TODO: Fix the division bug
# FIXME: Add bounds checking to get_item
''')
yield project
# =============================================================================
# E2E Test Fixtures
# =============================================================================
@pytest.fixture
async def live_client(api_url):
"""HTTP client for E2E tests against running server."""
async with AsyncClient(base_url=api_url, timeout=30.0) as client:
yield client
# =============================================================================
# Helper Functions
# =============================================================================
def assert_contains_any(text: str, substrings: list[str], case_sensitive: bool = False) -> bool:
"""Assert that text contains at least one of the substrings."""
check_text = text if case_sensitive else text.lower()
check_subs = substrings if case_sensitive else [s.lower() for s in substrings]
found = [s for s in check_subs if s in check_text]
assert found, f"Expected text to contain one of {substrings}, but none found in: {text[:200]}..."
return True
def assert_contains_all(text: str, substrings: list[str], case_sensitive: bool = False) -> bool:
"""Assert that text contains all of the substrings."""
check_text = text if case_sensitive else text.lower()
check_subs = substrings if case_sensitive else [s.lower() for s in substrings]
missing = [s for s in check_subs if s not in check_text]
assert not missing, f"Expected text to contain all of {substrings}, missing: {missing}"
return True
+225
View File
@@ -0,0 +1,225 @@
"""
End-to-end tests for the API.
These tests require the API server to be running and are skipped by default.
Run with: pytest tests/test_e2e.py -v --run-e2e
Start the server first: ./wakeup.sh
"""
import pytest
from tests.conftest import assert_contains_any
# All tests in this module require --run-e2e
pytestmark = [pytest.mark.e2e, pytest.mark.slow]
class TestHealthEndpoint:
"""E2E tests for health endpoint."""
@pytest.mark.anyio
async def test_health_check(self, live_client):
"""Test that health endpoint responds."""
response = await live_client.get("/health")
assert response.status_code == 200
data = response.json()
assert data.get("status") == "healthy"
class TestAgentEndpoints:
"""E2E tests for agent endpoints."""
@pytest.mark.anyio
async def test_list_agents(self, live_client):
"""Test listing agents via live API."""
response = await live_client.get("/agents/")
assert response.status_code == 200
data = response.json()
assert "agents" in data
assert len(data["agents"]) >= 1
# Verify explore agent exists
names = [a["name"] for a in data["agents"]]
assert "explore" in names
@pytest.mark.anyio
async def test_get_agent_info(self, live_client):
"""Test getting agent info via live API."""
response = await live_client.get("/agents/explore")
assert response.status_code == 200
data = response.json()
assert data["name"] == "explore"
assert "description" in data
@pytest.mark.anyio
async def test_run_agent(self, live_client, sample_project):
"""Test running agent via live API."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "explore",
"prompt": "List all files in this directory",
"working_dir": str(sample_project),
},
timeout=60.0, # LLM calls can be slow
)
assert response.status_code == 200
data = response.json()
assert data.get("success") is True
assert "response" in data
assert len(data["response"]) > 0
class TestStreamingEndpoint:
"""E2E tests for streaming endpoint."""
@pytest.mark.anyio
async def test_stream_agent(self, live_client, sample_project):
"""Test streaming agent responses via live API."""
async with live_client.stream(
"POST",
"/agents/stream",
json={
"agent_type": "explore",
"prompt": "List all Python files",
"working_dir": str(sample_project),
},
timeout=60.0,
) as response:
assert response.status_code == 200
assert response.headers.get("content-type") == "text/event-stream; charset=utf-8"
# Collect chunks
chunks = []
async for line in response.aiter_lines():
if line.startswith("data: "):
chunks.append(line)
# Should receive some data
assert len(chunks) >= 1
@pytest.mark.anyio
async def test_stream_invalid_agent(self, live_client):
"""Test streaming with invalid agent type."""
response = await live_client.post(
"/agents/stream",
json={
"agent_type": "nonexistent",
"prompt": "test",
"working_dir": ".",
},
)
assert response.status_code == 400
class TestErrorHandling:
"""E2E tests for error handling."""
@pytest.mark.anyio
async def test_invalid_agent_type(self, live_client):
"""Test error response for invalid agent."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "nonexistent",
"prompt": "test",
"working_dir": ".",
},
)
assert response.status_code == 400
data = response.json()
assert "detail" in data
@pytest.mark.anyio
async def test_missing_required_fields(self, live_client):
"""Test validation error for missing fields."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "explore",
# Missing prompt
},
)
assert response.status_code == 422
@pytest.mark.anyio
async def test_agent_not_found(self, live_client):
"""Test 404 for unknown agent info."""
response = await live_client.get("/agents/unknown_agent")
assert response.status_code == 404
class TestRealWorldScenarios:
"""E2E tests for real-world usage scenarios."""
@pytest.mark.anyio
async def test_explore_codebase(self, live_client, sample_project):
"""Test exploring a real codebase."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "explore",
"prompt": "What functions are defined in the src directory?",
"working_dir": str(sample_project),
},
timeout=60.0,
)
assert response.status_code == 200
data = response.json()
assert data.get("success") is True
# Should mention some functions
assert_contains_any(
data.get("response", ""),
["greet", "add", "divide", "factorial", "function"]
)
@pytest.mark.anyio
async def test_find_bugs(self, live_client, sample_project_with_bug):
"""Test finding bugs in code."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "explore",
"prompt": "Review buggy.py and identify any potential bugs or issues.",
"working_dir": str(sample_project_with_bug),
},
timeout=60.0,
)
assert response.status_code == 200
data = response.json()
assert data.get("success") is True
# Should identify issues
assert_contains_any(
data.get("response", ""),
["zero", "division", "bug", "error", "issue", "check"]
)
@pytest.mark.anyio
async def test_read_and_summarize(self, live_client, sample_project):
"""Test reading and summarizing a file."""
response = await live_client.post(
"/agents/run",
json={
"agent_type": "explore",
"prompt": "Read README.md and give me a one-sentence summary.",
"working_dir": str(sample_project),
},
timeout=60.0,
)
assert response.status_code == 200
data = response.json()
assert data.get("success") is True
assert len(data.get("response", "")) > 10
+181
View File
@@ -0,0 +1,181 @@
"""
Integration tests with real LLM.
These tests require Ollama to be running and are skipped by default.
Run with: pytest tests/test_integration.py -v --run-integration
Note: These tests are slow (each takes 5-30 seconds depending on LLM response time).
"""
import pytest
from tests.conftest import assert_contains_any
# All tests in this module require --run-integration
pytestmark = [pytest.mark.integration, pytest.mark.slow]
class TestExploreAgentIntegration:
"""Integration tests for the explore agent with real LLM."""
@pytest.fixture
def agent(self):
"""Get the explore agent."""
from src.domains.agents.explore.agent import explore_agent
return explore_agent
@pytest.mark.anyio
async def test_agent_can_list_files(self, agent, sample_project):
"""Test that agent can use glob to list files."""
result = await agent.run(
"List all Python files in this project. Just list the filenames.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Agent should mention the Python files
assert_contains_any(result, ["main.py", "utils.py", ".py"])
@pytest.mark.anyio
async def test_agent_can_read_file(self, agent, sample_project):
"""Test that agent can read file contents."""
result = await agent.run(
"Use the read_file tool to read src/main.py and list what functions are defined.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Agent should either mention functions or indicate it read the file
# LLMs can be unpredictable, so we check for various valid responses
assert_contains_any(result, [
"greet", "add", "divide", # Function names
"function", "def", # Generic function mentions
"main.py", # File reference
])
@pytest.mark.anyio
async def test_agent_can_search_content(self, agent, sample_project):
"""Test that agent can grep for patterns."""
result = await agent.run(
"Search for all TODO and FIXME comments in the codebase.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Should find no TODOs in the clean sample project
# (or correctly report none found)
assert result is not None and len(result) > 0
@pytest.mark.anyio
async def test_agent_finds_bugs(self, agent, sample_project_with_bug):
"""Test that agent can identify potential bugs."""
result = await agent.run(
"Review buggy.py and identify potential bugs or issues.",
working_dir=str(sample_project_with_bug),
allowed_paths=[str(sample_project_with_bug)],
)
# Agent should identify at least one issue
assert_contains_any(result, [
"zero", "division", "error", "bug", "issue",
"index", "bounds", "check", "validation"
])
@pytest.mark.anyio
async def test_agent_respects_path_restrictions(self, agent, sample_project):
"""Test that agent cannot access files outside allowed paths."""
result = await agent.run(
"Try to read the file /etc/passwd and show its contents.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Agent should not be able to read /etc/passwd
# Should not contain actual passwd file content
assert "root:x:0:0" not in result
class TestExploreAgentToolUsage:
"""Test that the agent correctly uses tools."""
@pytest.fixture
def agent(self):
"""Get the explore agent."""
from src.domains.agents.explore.agent import explore_agent
return explore_agent
@pytest.mark.anyio
async def test_agent_uses_glob_for_file_search(self, agent, sample_project):
"""Test that agent uses glob when searching for files."""
result = await agent.run(
"What markdown files exist in this project?",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Should find README.md
assert_contains_any(result, ["readme", "README.md", ".md"])
@pytest.mark.anyio
async def test_agent_uses_grep_for_content_search(self, agent, sample_project):
"""Test that agent uses grep for content search."""
result = await agent.run(
"Find where the 'factorial' function is defined and show its implementation.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Should find factorial in utils.py
assert_contains_any(result, ["factorial", "recursive", "utils"])
@pytest.mark.anyio
async def test_agent_reads_readme(self, agent, sample_project):
"""Test that agent can read and summarize README."""
result = await agent.run(
"Read the README.md and summarize what this project does.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Should understand the project from README
assert_contains_any(result, ["project", "python", "testing", "greeting", "math"])
@pytest.mark.anyio
async def test_agent_understands_project_structure(self, agent, sample_project):
"""Test that agent can understand project structure."""
result = await agent.run(
"Describe the directory structure of this project.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
)
# Should identify key directories
assert_contains_any(result, ["src", "tests", "directory", "folder", "structure"])
class TestAgentStreaming:
"""Test agent streaming functionality."""
@pytest.fixture
def agent(self):
"""Get the explore agent."""
from src.domains.agents.explore.agent import explore_agent
return explore_agent
@pytest.mark.anyio
async def test_agent_can_stream(self, agent, sample_project):
"""Test that agent streaming works."""
chunks = []
async for chunk in agent.run_stream(
"List the Python files in this project.",
working_dir=str(sample_project),
allowed_paths=[str(sample_project)],
):
chunks.append(chunk)
# Should receive at least one chunk
assert len(chunks) >= 1
# Combined result should mention files
full_result = "".join(chunks)
assert len(full_result) > 0