97 findings to zero. Most were mechanical — 52 unsorted import blocks, 10 unsorted __all__, assorted pyupgrade and simplify hints. Two were not, and both were visible only because the lint made me look. `webber version` did not exist. src/cli/commands/version.py defines show_version(), main.py imported it, and the registration line was never written — the CLI exposed chat and explore only. The import carried `# noqa: F401`, which is what kept the omission quiet: someone marked the symptom as intentional instead of asking why it was unused. show_version is not redundant with the --version flag; it prints the resolved Ollama URL, model and debug state, which is the form worth having when something is misconfigured. Registered, and the suppression dropped because the import is now genuinely used. test_spawn_explore_agent asserted nothing. It built a mock RunContext, patched get_agent, and stopped at the comment "For now, verify the explore agent would be called correctly". It had been counted as a passing test. An AST sweep of all 238 test functions found it was the only one, which is worth knowing — the problem was contained, not systemic. It is now skipped with a reason, so it reports as unfinished rather than as passing. Reducing it rather than deleting its imports was the point: tidying the imports would have made a hollow test look clean. Two findings were false positives, and both are recorded rather than silently worked around: B023 flagged run_agent closing over full_prompt and ctx. Traced: agent_task is awaited at line 326 before `continue` reaches the next iteration, so neither name can be rebound while the closure is pending, and the exception path cancels and awaits too. Not a bug. Bound as defaults anyway, because that stays true if the await ever moves. I had called it a live bug before tracing it, which is the mistake Rule 5 exists for. RUF012 flagged `rules: list[ApprovalRule] = []` on ApprovalRuleSet. Its suggested fix — annotate ClassVar — would remove the field from the model. ApprovalRuleSet is a pydantic model and pydantic deep-copies defaults per instance; verified by constructing two and confirming their lists are distinct objects. Suppressed with that evidence in the comment. Ruff cannot see the pydantic base because BaseSchema is a local subclass of BaseModel. Also moved a stray `from src.shared.logging import ...` that had drifted below a function definition, and merged a nested if in the ollama provider. 215 passed, 23 skipped, unchanged except for the new skip. `webber version` exercised end to end. mypy is NOT addressed here and the gate still fails on it — 55 errors in 14 files, 35 of them no-any-return from pydantic_ai's untyped returns. That was hidden behind ruff, because the gate stops at the first failing stage. Co-Authored-By: Claude <noreply@anthropic.com>
247 lines
8.4 KiB
Python
247 lines
8.4 KiB
Python
"""
|
|
Tests for the Task agent.
|
|
|
|
Tests registration, API endpoints, tool access, and spawn_agent functionality.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from src.domains.agents.base import get_agent, list_agents
|
|
from src.domains.agents.task import TaskAgentImpl, task_agent
|
|
|
|
|
|
class TestTaskAgentRegistration:
|
|
"""Tests for Task agent registration."""
|
|
|
|
def test_task_agent_registered(self):
|
|
"""Test that task agent is registered in registry."""
|
|
agent = get_agent("task")
|
|
assert agent is not None
|
|
assert agent.name == "task"
|
|
|
|
def test_task_agent_in_list(self):
|
|
"""Test that task agent appears in agent list."""
|
|
agents = list_agents()
|
|
names = [a["name"] for a in agents]
|
|
assert "task" in names
|
|
|
|
def test_task_agent_has_description(self):
|
|
"""Test that task agent has a description."""
|
|
agent = get_agent("task")
|
|
assert agent is not None
|
|
assert len(agent.description) > 0
|
|
assert "task" in agent.description.lower() or "autonomous" in agent.description.lower()
|
|
|
|
def test_task_agent_singleton(self):
|
|
"""Test that task_agent is the registered instance."""
|
|
registered = get_agent("task")
|
|
assert registered is task_agent
|
|
|
|
def test_task_agent_is_correct_type(self):
|
|
"""Test that task agent is correct implementation type."""
|
|
assert isinstance(task_agent, TaskAgentImpl)
|
|
|
|
|
|
class TestTaskAgentTools:
|
|
"""Tests for Task agent tool access."""
|
|
|
|
def test_task_agent_has_all_tools(self):
|
|
"""Test that task agent has all 9 tools."""
|
|
agent = task_agent.agent
|
|
tool_names = list(agent._function_toolset.tools.keys())
|
|
|
|
# Should have 9 tools total
|
|
assert len(tool_names) == 9
|
|
|
|
def test_task_agent_has_read_only_tools(self):
|
|
"""Test that task agent has read-only tools."""
|
|
agent = task_agent.agent
|
|
tool_names = list(agent._function_toolset.tools.keys())
|
|
|
|
assert "read_file" in tool_names
|
|
assert "glob_files" in tool_names
|
|
assert "grep_content" in tool_names
|
|
assert "bash_readonly" in tool_names
|
|
|
|
def test_task_agent_has_write_tools(self):
|
|
"""Test that task agent has write tools."""
|
|
agent = task_agent.agent
|
|
tool_names = list(agent._function_toolset.tools.keys())
|
|
|
|
assert "edit_file" in tool_names
|
|
assert "write_file" in tool_names
|
|
assert "bash" in tool_names
|
|
|
|
def test_task_agent_has_external_tools(self):
|
|
"""Test that task agent has external tools."""
|
|
agent = task_agent.agent
|
|
tool_names = list(agent._function_toolset.tools.keys())
|
|
|
|
assert "web_search" in tool_names
|
|
|
|
def test_task_agent_has_spawn_agent_tool(self):
|
|
"""Test that task agent has spawn_agent orchestration tool."""
|
|
agent = task_agent.agent
|
|
tool_names = list(agent._function_toolset.tools.keys())
|
|
|
|
assert "spawn_agent" in tool_names
|
|
|
|
|
|
class TestSpawnAgentTool:
|
|
"""Tests for spawn_agent orchestration functionality."""
|
|
|
|
@pytest.mark.skip(
|
|
reason="never finished — the body built a mock context and then asserted "
|
|
"nothing, so it counted as a passing test while verifying nothing"
|
|
)
|
|
@pytest.mark.anyio
|
|
async def test_spawn_explore_agent(self):
|
|
"""Spawning an explore agent should delegate to the explore agent.
|
|
|
|
The scaffolding that used to sit here — a MagicMock RunContext, an
|
|
AgentContext with a /tmp working dir, and a patch of
|
|
src.domains.agents.base.get_agent — ran and then stopped at the comment
|
|
"For now, verify the explore agent would be called correctly". There was
|
|
no assertion, so it passed unconditionally.
|
|
|
|
Removed rather than tidied: ruff flagged its imports as unused, and
|
|
deleting those would have made the test look clean while leaving it
|
|
hollow. git history has the setup for whoever finishes this.
|
|
"""
|
|
|
|
@pytest.mark.anyio
|
|
async def test_spawn_unknown_agent_returns_error(self):
|
|
"""Test that spawning unknown agent type returns error."""
|
|
|
|
# We can't easily test the tool directly, but we can verify
|
|
# the agent type validation logic
|
|
allowed_types = ["explore", "plan"]
|
|
assert "nonexistent" not in allowed_types
|
|
assert "task" not in allowed_types # Task should be blocked
|
|
|
|
def test_spawn_task_agent_blocked(self):
|
|
"""Test that spawning nested task agents is blocked."""
|
|
# Verify the validation logic prevents recursion
|
|
# The spawn_agent tool should return an error for agent_type="task"
|
|
allowed_types = ["explore", "plan"]
|
|
assert "task" not in allowed_types
|
|
|
|
|
|
class TestTaskAgentAPI:
|
|
"""Tests for Task agent REST API."""
|
|
|
|
@pytest.mark.anyio
|
|
async def test_list_agents_includes_task(self, auth_client):
|
|
"""Test that agent list includes task agent."""
|
|
response = await auth_client.get("/agents/")
|
|
|
|
assert response.status_code == 200
|
|
data = response.json()
|
|
names = [a["name"] for a in data["agents"]]
|
|
assert "task" in names
|
|
|
|
@pytest.mark.anyio
|
|
async def test_get_task_agent_info(self, auth_client):
|
|
"""Test getting task agent info."""
|
|
response = await auth_client.get("/agents/task")
|
|
|
|
assert response.status_code == 200
|
|
data = response.json()
|
|
assert data["name"] == "task"
|
|
assert "description" in data
|
|
assert len(data["description"]) > 0
|
|
|
|
@pytest.mark.anyio
|
|
async def test_run_task_with_invalid_body(self, auth_client):
|
|
"""Test running task agent with invalid request."""
|
|
response = await auth_client.post(
|
|
"/agents/run",
|
|
json={
|
|
"agent_type": "task",
|
|
# Missing prompt
|
|
}
|
|
)
|
|
|
|
assert response.status_code == 422
|
|
|
|
@pytest.mark.anyio
|
|
async def test_stream_task_with_invalid_body(self, auth_client):
|
|
"""Test streaming task agent with invalid request."""
|
|
response = await auth_client.post(
|
|
"/agents/stream",
|
|
json={
|
|
"agent_type": "task",
|
|
# Missing prompt
|
|
}
|
|
)
|
|
|
|
assert response.status_code == 422
|
|
|
|
|
|
class TestTaskAgentProperties:
|
|
"""Tests for Task agent properties and configuration."""
|
|
|
|
def test_task_agent_name(self):
|
|
"""Test task agent name property."""
|
|
assert task_agent.name == "task"
|
|
|
|
def test_task_agent_description_not_empty(self):
|
|
"""Test task agent description is not empty."""
|
|
assert task_agent.description
|
|
assert len(task_agent.description) > 10
|
|
|
|
def test_task_agent_creates_agent_lazily(self):
|
|
"""Test that PydanticAI agent is created lazily."""
|
|
# Create a fresh instance
|
|
fresh_agent = TaskAgentImpl()
|
|
|
|
# _agents dict should be empty before first access
|
|
assert len(fresh_agent._agents) == 0
|
|
|
|
# Access the agent property (creates default mode agent)
|
|
_ = fresh_agent.agent
|
|
|
|
# Now _agents should have one entry
|
|
assert len(fresh_agent._agents) == 1
|
|
|
|
|
|
class TestAllAgentsRegistered:
|
|
"""Tests to verify all three agents are registered."""
|
|
|
|
def test_all_agents_in_registry(self):
|
|
"""Test that explore, plan, and task agents are all registered."""
|
|
agents = list_agents()
|
|
names = [a["name"] for a in agents]
|
|
|
|
assert "explore" in names
|
|
assert "plan" in names
|
|
assert "task" in names
|
|
assert len(names) == 3
|
|
|
|
def test_agent_hierarchy(self):
|
|
"""Test the agent capability hierarchy."""
|
|
explore = get_agent("explore")
|
|
plan = get_agent("plan")
|
|
task = get_agent("task")
|
|
|
|
explore_tools = list(explore.agent._function_toolset.tools.keys())
|
|
plan_tools = list(plan.agent._function_toolset.tools.keys())
|
|
task_tools = list(task.agent._function_toolset.tools.keys())
|
|
|
|
# Explore has all tools (read + write)
|
|
assert "edit_file" in explore_tools
|
|
assert "write_file" in explore_tools
|
|
|
|
# Plan has read-only tools
|
|
assert "edit_file" not in plan_tools
|
|
assert "write_file" not in plan_tools
|
|
|
|
# Task has all tools plus spawn_agent
|
|
assert "edit_file" in task_tools
|
|
assert "write_file" in task_tools
|
|
assert "spawn_agent" in task_tools
|
|
|
|
# Only Task has spawn_agent
|
|
assert "spawn_agent" not in explore_tools
|
|
assert "spawn_agent" not in plan_tools
|