Compare commits

...
3 Commits
Author SHA1 Message Date
jpmschweitzerandClaude Opus 4.5 583c407edd fix: Redis bool storage, tool tracking matching, e2e fixture scope
Build and Push / build (release) Successful in 53s
- Convert booleans to strings for Redis hset (Redis doesn't accept bool)
- Extract capability from delegate_to_X tool names for tracking
- Use loop_scope="module" for pytest-asyncio module-scoped fixtures
- Add note about using venv for tests in AGENTS.md

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-16 14:53:51 +01:00
jpmschweitzer 404e8fc106 add pre deploy check 2025-12-16 09:36:17 +01:00
jpmschweitzerandClaude Opus 4.5 54a27b481a docs: add release flow section to AGENTS.md
Documents the version bump, changelog update, tagging, and
deployment verification steps.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-16 09:33:52 +01:00
8 changed files with 185 additions and 25 deletions
+31
View File
@@ -22,6 +22,11 @@ This document contains instructions and documentation references for AI assistan
* **Test REST endpoints** against `http://localhost:8777` using curl or similar tools * **Test REST endpoints** against `http://localhost:8777` using curl or similar tools
* **Only deploy** when a phase or feature is complete and tested locally * **Only deploy** when a phase or feature is complete and tested locally
* **Environment**: Copy `.env.example` to `.env` and configure for your local setup (Ollama, Redis, Qdrant hosts) * **Environment**: Copy `.env.example` to `.env` and configure for your local setup (Ollama, Redis, Qdrant hosts)
* **Running tests**: Always use the venv explicitly to avoid environment mismatches:
```bash
.venv/bin/python -m pytest tests/ # All tests
.venv/bin/python -m pytest tests/core/ -v # Core tests only
```
### 🌐 Internal Service Access ### 🌐 Internal Service Access
* **git.schweitz.net**: Access via `http://localhost:3002` (direct Gitea) to bypass Authentik SSO * **git.schweitz.net**: Access via `http://localhost:3002` (direct Gitea) to bypass Authentik SSO
@@ -51,6 +56,32 @@ This document contains instructions and documentation references for AI assistan
* **Update `CHANGELOG.md`** with every user-facing change. * **Update `CHANGELOG.md`** with every user-facing change.
* Format: `## [Unreleased] - YYYY-MM-DD` followed by `### Added`, `### Changed`, or `### Fixed`. * Format: `## [Unreleased] - YYYY-MM-DD` followed by `### Added`, `### Changed`, or `### Fixed`.
### 🚀 Release Flow
When changes are ready for deployment:
1. **Ask user if deploy cycle is desired**
2. **Update version** in `pyproject.toml`:
- Bug fixes: bump patch version (1.8.3 → 1.8.4)
- New features: bump minor version (1.8.4 → 1.9.0)
3. **Update CHANGELOG.md**:
- Move items from `[Unreleased]` to new version section
- Add release date: `## [1.8.4] - 2025-12-16`
4. **Commit and tag**:
```bash
git add -A
git commit -m "fix: description of changes"
git tag v1.8.4
git push origin main --tags
```
5. **CI/CD triggers automatically**:
- Gitea CI builds Docker image on new tag
- Watchtower pulls and deploys to production
- Verify deployment: `curl http://192.168.86.149:8000/health`
--- ---
## 2. FastAPI Architecture & Best Practices ## 2. FastAPI Architecture & Best Practices
+8
View File
@@ -7,6 +7,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [Unreleased] ## [Unreleased]
## [1.8.5] - 2025-12-16
### Fixed
- **Redis benchmark boolean storage** - Convert booleans to strings for Redis `hset` (Redis doesn't accept bool type directly)
- **Tool tracking capability matching** - `delegate_to_librarian` now correctly recognized as using "librarian" capability when checking Steward recommendations
- **E2E test fixture scope** - Fixed pytest-asyncio ScopeMismatch error by using `loop_scope="module"` for module-scoped async fixtures
## [1.8.4] - 2025-12-16 ## [1.8.4] - 2025-12-16
### Fixed ### Fixed
+1 -1
View File
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project] [project]
name = "tatlock" name = "tatlock"
version = "1.8.4" version = "1.8.5"
description = "OpenAI-compatible API with Ollama backend" description = "OpenAI-compatible API with Ollama backend"
requires-python = ">=3.12" requires-python = ">=3.12"
dependencies = [] dependencies = []
+8
View File
@@ -47,6 +47,10 @@ class PerformanceBenchmark(BaseModel):
data = self.model_dump() data = self.model_dump()
data["timestamp"] = self.timestamp.isoformat() data["timestamp"] = self.timestamp.isoformat()
data["metadata"] = json.dumps(self.metadata) data["metadata"] = json.dumps(self.metadata)
# Convert booleans to strings (Redis doesn't accept bool type)
for key, value in data.items():
if isinstance(value, bool):
data[key] = str(value)
return data return data
@classmethod @classmethod
@@ -54,6 +58,10 @@ class PerformanceBenchmark(BaseModel):
"""Reconstruct from Redis dict.""" """Reconstruct from Redis dict."""
data["timestamp"] = datetime.fromisoformat(data["timestamp"]) data["timestamp"] = datetime.fromisoformat(data["timestamp"])
data["metadata"] = json.loads(data.get("metadata", "{}")) data["metadata"] = json.loads(data.get("metadata", "{}"))
# Convert string booleans back to bool
for key in ["success", "was_recommended", "was_actually_used"]:
if key in data and isinstance(data[key], str):
data[key] = data[key] == "True"
return cls(**data) return cls(**data)
+25 -6
View File
@@ -43,6 +43,16 @@ class ToolCallTracker:
conversation_id=conversation_id, conversation_id=conversation_id,
) )
def _extract_capability(self, tool_name: str) -> str:
"""
Extract capability name from tool name.
Tool names like 'delegate_to_librarian' map to capability 'librarian'.
"""
if tool_name.startswith("delegate_to_"):
return tool_name.replace("delegate_to_", "")
return tool_name
async def track_call(self, tool_name: str, duration: float): async def track_call(self, tool_name: str, duration: float):
""" """
Record a tool call with timing. Record a tool call with timing.
@@ -56,8 +66,9 @@ class ToolCallTracker:
self.actual_calls[tool_name] = [] self.actual_calls[tool_name] = []
self.actual_calls[tool_name].append(duration) self.actual_calls[tool_name].append(duration)
# Check if tool was recommended # Check if tool was recommended (normalize tool name to capability)
was_recommended = tool_name in self.recommended_capabilities capability = self._extract_capability(tool_name)
was_recommended = capability in self.recommended_capabilities
if not was_recommended: if not was_recommended:
logger.warning( logger.warning(
@@ -98,8 +109,12 @@ class ToolCallTracker:
Called after Tatlock completes its response to identify Called after Tatlock completes its response to identify
tools that were recommended but never used. tools that were recommended but never used.
""" """
# Normalize actual tool names to capabilities for comparison
used_capabilities = {
self._extract_capability(tool) for tool in self.actual_calls.keys()
}
# Find tools that were recommended but not used # Find tools that were recommended but not used
unused_tools = self.recommended_capabilities - set(self.actual_calls.keys()) unused_tools = self.recommended_capabilities - used_capabilities
if unused_tools: if unused_tools:
logger.info( logger.info(
@@ -145,7 +160,11 @@ class ToolCallTracker:
Dict with tracking statistics Dict with tracking statistics
""" """
total_calls = sum(len(durations) for durations in self.actual_calls.values()) total_calls = sum(len(durations) for durations in self.actual_calls.values())
unused = self.recommended_capabilities - set(self.actual_calls.keys()) # Normalize actual tool names to capabilities for comparison
used_capabilities = {
self._extract_capability(tool) for tool in self.actual_calls.keys()
}
unused = self.recommended_capabilities - used_capabilities
return { return {
"recommended_capabilities": list(self.recommended_capabilities), "recommended_capabilities": list(self.recommended_capabilities),
@@ -154,11 +173,11 @@ class ToolCallTracker:
"total_calls": total_calls, "total_calls": total_calls,
"accuracy": { "accuracy": {
"recommended_and_used": len( "recommended_and_used": len(
self.recommended_capabilities & set(self.actual_calls.keys()) self.recommended_capabilities & used_capabilities
), ),
"recommended_but_unused": len(unused), "recommended_but_unused": len(unused),
"not_recommended_but_used": len( "not_recommended_but_used": len(
set(self.actual_calls.keys()) - self.recommended_capabilities used_capabilities - self.recommended_capabilities
), ),
}, },
} }
+9 -8
View File
@@ -61,7 +61,7 @@ class TestPerformanceBenchmark:
redis_dict = benchmark.to_redis_dict() redis_dict = benchmark.to_redis_dict()
assert redis_dict["operation"] == "test_op" assert redis_dict["operation"] == "test_op"
assert redis_dict["duration_seconds"] == 1.0 assert redis_dict["duration_seconds"] == 1.0
assert redis_dict["success"] is True assert redis_dict["success"] == "True" # Booleans stored as strings in Redis
assert isinstance(redis_dict["timestamp"], str) assert isinstance(redis_dict["timestamp"], str)
assert isinstance(redis_dict["metadata"], str) assert isinstance(redis_dict["metadata"], str)
@@ -72,7 +72,7 @@ class TestPerformanceBenchmark:
"timestamp": now.isoformat(), "timestamp": now.isoformat(),
"operation": "test_op", "operation": "test_op",
"duration_seconds": 1.5, "duration_seconds": 1.5,
"success": True, "success": "True", # Booleans stored as strings in Redis
"metadata": json.dumps({"test": "data"}), "metadata": json.dumps({"test": "data"}),
"recommendation_count": None, "recommendation_count": None,
"confidence": None, "confidence": None,
@@ -85,6 +85,7 @@ class TestPerformanceBenchmark:
benchmark = PerformanceBenchmark.from_redis_dict(redis_dict) benchmark = PerformanceBenchmark.from_redis_dict(redis_dict)
assert benchmark.operation == "test_op" assert benchmark.operation == "test_op"
assert benchmark.duration_seconds == 1.5 assert benchmark.duration_seconds == 1.5
assert benchmark.success is True # Converted back to bool
assert benchmark.metadata == {"test": "data"} assert benchmark.metadata == {"test": "data"}
@@ -162,12 +163,12 @@ class TestBenchmarkStore:
mock_key = f"benchmark:test_op:{int(now.timestamp() * 1000)}" mock_key = f"benchmark:test_op:{int(now.timestamp() * 1000)}"
mock_redis.zrevrangebyscore.return_value = [mock_key] mock_redis.zrevrangebyscore.return_value = [mock_key]
# Mock hgetall to return proper data # Mock hgetall to return proper data (booleans as strings, like Redis)
mock_redis.hgetall.return_value = { mock_redis.hgetall.return_value = {
"timestamp": now.isoformat(), "timestamp": now.isoformat(),
"operation": "test_op", "operation": "test_op",
"duration_seconds": 1.5, # Numeric, not string "duration_seconds": 1.5, # Numeric, not string
"success": True, "success": "True", # Booleans stored as strings in Redis
"metadata": "{}", "metadata": "{}",
"recommendation_count": None, "recommendation_count": None,
"confidence": None, "confidence": None,
@@ -237,7 +238,7 @@ class TestBenchmarkStore:
"timestamp": now.isoformat(), "timestamp": now.isoformat(),
"operation": "test_op", "operation": "test_op",
"duration_seconds": float(data["duration_seconds"]), "duration_seconds": float(data["duration_seconds"]),
"success": data["success"] == "True", "success": data["success"], # Pass string through, from_redis_dict converts
"metadata": "{}", "metadata": "{}",
"recommendation_count": None, "recommendation_count": None,
"confidence": None, "confidence": None,
@@ -296,14 +297,14 @@ class TestBenchmarkStore:
"timestamp": now.isoformat(), "timestamp": now.isoformat(),
"operation": "tool_call", "operation": "tool_call",
"duration_seconds": 1.0, "duration_seconds": 1.0,
"success": True, "success": "True", # Booleans stored as strings in Redis
"metadata": "{}", "metadata": "{}",
"recommendation_count": None, "recommendation_count": None,
"confidence": None, "confidence": None,
"tool_name": "test_tool", "tool_name": "test_tool",
"conversation_id": None, "conversation_id": None,
"was_recommended": data["was_recommended"] == "True", "was_recommended": data["was_recommended"], # Already strings
"was_actually_used": data["was_actually_used"] == "True", "was_actually_used": data["was_actually_used"], # Already strings
} }
mock_redis.hgetall.side_effect = mock_hgetall mock_redis.hgetall.side_effect = mock_hgetall
+101
View File
@@ -0,0 +1,101 @@
"""
Tests for tool call tracking.
Tests capability extraction and recommendation matching.
"""
from unittest.mock import AsyncMock, patch
import pytest
from src.core.tool_tracking import ToolCallTracker
class TestToolCallTracker:
"""Test ToolCallTracker functionality."""
def test_extract_capability_delegation_tool(self):
"""Test extracting capability from delegation tool name."""
tracker = ToolCallTracker(recommended_capabilities=["librarian"])
assert tracker._extract_capability("delegate_to_librarian") == "librarian"
assert tracker._extract_capability("delegate_to_biographer") == "biographer"
assert tracker._extract_capability("delegate_to_housekeeper") == "housekeeper"
def test_extract_capability_non_delegation_tool(self):
"""Test that non-delegation tools return unchanged."""
tracker = ToolCallTracker(recommended_capabilities=[])
assert tracker._extract_capability("calculate") == "calculate"
assert tracker._extract_capability("search_web") == "search_web"
@pytest.mark.asyncio
async def test_track_call_recognizes_delegation_as_recommended(self):
"""Test that delegate_to_X is recognized when X is recommended."""
tracker = ToolCallTracker(
recommended_capabilities=["librarian", "biographer"]
)
with patch("src.core.tool_tracking.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
await tracker.track_call("delegate_to_librarian", 1.0)
# Should NOT log warning since librarian was recommended
call_args = mock_store.return_value.record.call_args
benchmark = call_args[0][0]
assert benchmark.was_recommended is True
@pytest.mark.asyncio
async def test_track_call_detects_not_recommended(self):
"""Test that unrecommended tools are flagged."""
tracker = ToolCallTracker(
recommended_capabilities=["librarian"]
)
with patch("src.core.tool_tracking.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
await tracker.track_call("delegate_to_housekeeper", 1.0)
call_args = mock_store.return_value.record.call_args
benchmark = call_args[0][0]
assert benchmark.was_recommended is False
def test_get_summary_with_delegation_tools(self):
"""Test summary correctly maps delegation tools to capabilities."""
tracker = ToolCallTracker(
recommended_capabilities=["librarian", "biographer"]
)
tracker.actual_calls = {
"delegate_to_librarian": [1.0, 2.0],
"delegate_to_housekeeper": [0.5], # Not recommended
}
summary = tracker.get_summary()
assert summary["accuracy"]["recommended_and_used"] == 1 # librarian
assert summary["accuracy"]["recommended_but_unused"] == 1 # biographer
assert summary["accuracy"]["not_recommended_but_used"] == 1 # housekeeper
@pytest.mark.asyncio
async def test_finalize_with_delegation_tools(self):
"""Test finalize correctly identifies unused recommendations."""
tracker = ToolCallTracker(
recommended_capabilities=["librarian", "biographer"]
)
tracker.actual_calls = {
"delegate_to_librarian": [1.0],
}
with patch("src.core.tool_tracking.get_benchmark_store") as mock_store:
mock_store.return_value.record = AsyncMock()
await tracker.finalize()
# Should record benchmark for unused biographer
assert mock_store.return_value.record.called
call_args = mock_store.return_value.record.call_args
benchmark = call_args[0][0]
assert benchmark.tool_name == "biographer"
assert benchmark.was_recommended is True
assert benchmark.was_actually_used is False
+2 -10
View File
@@ -8,8 +8,8 @@ These tests hit the actual running server and test the full stack:
- Response formatting - Response formatting
""" """
import pytest import pytest
import pytest_asyncio
import httpx import httpx
import asyncio
from typing import AsyncGenerator from typing import AsyncGenerator
# Test server base URL (assumes server is running on localhost:8777 via ./wakeup.sh) # Test server base URL (assumes server is running on localhost:8777 via ./wakeup.sh)
@@ -17,15 +17,7 @@ BASE_URL = "http://localhost:8777"
API_TIMEOUT = 120.0 # 120 second timeout for LLM calls API_TIMEOUT = 120.0 # 120 second timeout for LLM calls
@pytest.fixture(scope="module") @pytest_asyncio.fixture(loop_scope="module", scope="module")
def event_loop():
"""Create event loop for async tests."""
loop = asyncio.get_event_loop_policy().new_event_loop()
yield loop
loop.close()
@pytest.fixture(scope="module")
async def client() -> AsyncGenerator[httpx.AsyncClient, None]: async def client() -> AsyncGenerator[httpx.AsyncClient, None]:
"""HTTP client for making requests.""" """HTTP client for making requests."""
async with httpx.AsyncClient(base_url=BASE_URL, timeout=API_TIMEOUT) as client: async with httpx.AsyncClient(base_url=BASE_URL, timeout=API_TIMEOUT) as client: