Compare commits
138
Commits
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"env": {
|
||||||
|
"PQL_VAULT": "/mnt/media/Projects/tatlock"
|
||||||
|
},
|
||||||
|
"permissions": {
|
||||||
|
"allow": [
|
||||||
|
"Bash(pql)",
|
||||||
|
"Bash(pql *)",
|
||||||
|
"Bash(/home/jpmschweitzer/.local/bin/pql:*)",
|
||||||
|
"Bash(git status:*)",
|
||||||
|
"Bash(git log:*)",
|
||||||
|
"Bash(git diff:*)",
|
||||||
|
"Bash(git branch:*)",
|
||||||
|
"Bash(make test:*)",
|
||||||
|
"Bash(make test-unit:*)",
|
||||||
|
"Bash(make test-contracts:*)",
|
||||||
|
"Bash(make lint:*)",
|
||||||
|
"Bash(make typecheck:*)",
|
||||||
|
"Bash(.venv/bin/python -m pytest:*)",
|
||||||
|
"Bash(.venv/bin/pytest:*)",
|
||||||
|
"Bash(pytest:*)",
|
||||||
|
"Bash(ruff check:*)",
|
||||||
|
"Bash(mypy:*)",
|
||||||
|
"Bash(docker logs tatlock:*)",
|
||||||
|
"Bash(curl -s http://localhost:8000/*)",
|
||||||
|
"Bash(curl -s http://localhost:8777/*)"
|
||||||
|
],
|
||||||
|
"deny": [
|
||||||
|
"Bash(/mnt/media/Projects/cladmin/ops/bin/toj)",
|
||||||
|
"Bash(/mnt/media/Projects/cladmin/ops/bin/toj:*)",
|
||||||
|
"Bash(chmod -R 777 *)",
|
||||||
|
"Bash(chmod 777 *)",
|
||||||
|
"Bash(dd if=*)",
|
||||||
|
"Bash(find * -delete*)",
|
||||||
|
"Bash(find * -exec*)",
|
||||||
|
"Bash(git * add --all*)",
|
||||||
|
"Bash(git * add -A*)",
|
||||||
|
"Bash(git * add .)",
|
||||||
|
"Bash(git * branch -D *)",
|
||||||
|
"Bash(git * checkout -- *)",
|
||||||
|
"Bash(git * clean -fd*)",
|
||||||
|
"Bash(git * clean -fdx*)",
|
||||||
|
"Bash(git * commit --no-verify*)",
|
||||||
|
"Bash(git * merge --no-ff*)",
|
||||||
|
"Bash(git * push --force*)",
|
||||||
|
"Bash(git * push -f*)",
|
||||||
|
"Bash(git * reset --hard*)",
|
||||||
|
"Bash(git * restore .*)",
|
||||||
|
"Bash(git add --all*)",
|
||||||
|
"Bash(git add -A*)",
|
||||||
|
"Bash(git add .)",
|
||||||
|
"Bash(git branch -D *)",
|
||||||
|
"Bash(git checkout -- *)",
|
||||||
|
"Bash(git clean -fd*)",
|
||||||
|
"Bash(git clean -fdx*)",
|
||||||
|
"Bash(git commit --no-verify*)",
|
||||||
|
"Bash(git merge --no-ff*)",
|
||||||
|
"Bash(git push --force*)",
|
||||||
|
"Bash(git push -f*)",
|
||||||
|
"Bash(git reset --hard*)",
|
||||||
|
"Bash(git restore .*)",
|
||||||
|
"Bash(mkfs*)",
|
||||||
|
"Bash(ollama rm *)",
|
||||||
|
"Bash(redis-cli * FLUSHALL*)",
|
||||||
|
"Bash(redis-cli * FLUSHDB*)",
|
||||||
|
"Bash(rm -rf $HOME)",
|
||||||
|
"Bash(rm -rf /)",
|
||||||
|
"Bash(rm -rf ~)",
|
||||||
|
"Bash(su *)",
|
||||||
|
"Bash(sudo *)",
|
||||||
|
"Bash(toj)",
|
||||||
|
"Bash(toj:*)"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
+40
-10
@@ -1,6 +1,5 @@
|
|||||||
# Application Configuration
|
# Application Configuration
|
||||||
APP_NAME="OpenAI-Compatible API"
|
APP_NAME="OpenAI-Compatible API"
|
||||||
APP_VERSION="0.1.0"
|
|
||||||
ENVIRONMENT=development
|
ENVIRONMENT=development
|
||||||
DEBUG=false
|
DEBUG=false
|
||||||
|
|
||||||
@@ -9,25 +8,56 @@ API_HOST=0.0.0.0
|
|||||||
API_PORT=8000
|
API_PORT=8000
|
||||||
API_PREFIX=/v1
|
API_PREFIX=/v1
|
||||||
|
|
||||||
# Ollama Configuration
|
# Ollama Configuration (local - primary backend)
|
||||||
OLLAMA_HOST=http://your-ollama-host:11434
|
OLLAMA_HOST=http://localhost:11434
|
||||||
OLLAMA_DEFAULT_MODEL=mistral-nemo:latest
|
OLLAMA_DEFAULT_MODEL=gemma4:e2b
|
||||||
OLLAMA_TIMEOUT=120
|
OLLAMA_TIMEOUT=120
|
||||||
|
STEWARD_TIMEOUT=60
|
||||||
|
|
||||||
|
# Anthropic Configuration (Claude - cloud fallback)
|
||||||
|
# Set ANTHROPIC_API_KEY to keep the Claude fallback available: it is used
|
||||||
|
# automatically when Ollama is down, or exclusively when PREFER_CLOUD_BACKEND=true
|
||||||
|
# Without an API key, Tatlock uses Ollama only
|
||||||
|
# ANTHROPIC_API_KEY=sk-ant-api03-your-key-here
|
||||||
|
ANTHROPIC_MODEL=claude-sonnet-5
|
||||||
|
PREFER_CLOUD_BACKEND=false
|
||||||
|
|
||||||
# SearXNG Configuration
|
# SearXNG Configuration
|
||||||
SEARXNG_HOST=http://searxng:8087
|
SEARXNG_HOST=http://localhost:8087
|
||||||
SEARXNG_TIMEOUT=30
|
SEARXNG_TIMEOUT=30
|
||||||
|
|
||||||
# Redis Configuration
|
# Redis Configuration
|
||||||
REDIS_HOST=redis-shared
|
REDIS_HOST=localhost
|
||||||
REDIS_PORT=6379
|
REDIS_PORT=6379
|
||||||
REDIS_DB=1
|
REDIS_MEMORY_DB=1
|
||||||
REDIS_TIMEOUT=5
|
REDIS_TIMEOUT=5
|
||||||
|
|
||||||
|
# Qdrant Configuration
|
||||||
|
QDRANT_HOST=localhost
|
||||||
|
QDRANT_PORT=6333
|
||||||
|
|
||||||
# Logging
|
# Logging
|
||||||
LOG_LEVEL=INFO
|
# LOG_LEVEL is auto-selected based on ENVIRONMENT if not set:
|
||||||
ENABLE_BENCHMARKS=true
|
# - development: DEBUG (maximum verbosity)
|
||||||
|
# - production: WARNING (minimal noise)
|
||||||
|
# Uncomment to override: LOG_LEVEL=INFO
|
||||||
# Note: Log format is auto-selected based on ENVIRONMENT (console for dev, json for production)
|
# Note: Log format is auto-selected based on ENVIRONMENT (console for dev, json for production)
|
||||||
|
|
||||||
|
# User Configuration
|
||||||
|
# DEFAULT_USER is auto-selected based on ENVIRONMENT if not set:
|
||||||
|
# - development/testing: llm_tester (isolated test scope)
|
||||||
|
# - production: jpmschweitzer (real user)
|
||||||
|
# Uncomment to override: DEFAULT_USER=your_username
|
||||||
|
|
||||||
|
# Library-Desk Configuration (The Librarian backend)
|
||||||
|
# LIBRARY_DESK_HOST=http://localhost:8089
|
||||||
|
# LIBRARY_DESK_API_KEY=your-library-desk-api-key
|
||||||
|
# LIBRARY_DESK_TIMEOUT=60
|
||||||
|
|
||||||
|
# Core-API Configuration (The Housekeeper backend)
|
||||||
|
# CORE_API_HOST=http://localhost:8090
|
||||||
|
# CORE_API_KEY=your-core-api-key
|
||||||
|
# CORE_API_TIMEOUT=30
|
||||||
|
|
||||||
# CORS (comma-separated list)
|
# CORS (comma-separated list)
|
||||||
CORS_ORIGINS=*
|
CORS_ORIGINS=["*"]
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
.pql/changelog/*.sql merge=union
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
name: Build and Push
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- 'v[0-9]*'
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
release:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Create Gitea Release
|
||||||
|
run: |
|
||||||
|
curl -sf -X POST \
|
||||||
|
-H "Authorization: token ${{ secrets.GITHUB_TOKEN }}" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{"tag_name": "${{ github.ref_name }}", "name": "Release ${{ github.ref_name }}", "body": "Automated release for ${{ github.ref_name }}"}' \
|
||||||
|
"${{ github.server_url }}/api/v1/repos/${{ github.repository }}/releases"
|
||||||
|
|
||||||
|
build:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Login to Gitea Registry
|
||||||
|
uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: git.schweitz.net
|
||||||
|
username: ${{ secrets.REGISTRY_USER }}
|
||||||
|
password: ${{ secrets.REGISTRY_PASSWORD }}
|
||||||
|
|
||||||
|
- name: Build and push
|
||||||
|
uses: docker/build-push-action@v6
|
||||||
|
with:
|
||||||
|
context: .
|
||||||
|
push: true
|
||||||
|
provenance: false
|
||||||
|
sbom: false
|
||||||
|
tags: |
|
||||||
|
git.schweitz.net/jpmschweitzer/tatlock:latest
|
||||||
|
git.schweitz.net/jpmschweitzer/tatlock:${{ github.ref_name }}
|
||||||
|
|
||||||
|
- name: Trigger Watchtower update
|
||||||
|
if: success()
|
||||||
|
run: |
|
||||||
|
curl -sf -H "Authorization: Bearer ${{ secrets.WATCHTOWER_TOKEN }}" \
|
||||||
|
http://watchtower:8080/v1/update
|
||||||
Executable
+13
@@ -0,0 +1,13 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Trigger only. The checks live in the Makefile, where they can be read, run by
|
||||||
|
# hand (`make pre-push`), and changed under review.
|
||||||
|
#
|
||||||
|
# This file is identical in every repo in this workspace, deliberately: the call
|
||||||
|
# surface is the same everywhere even though what each gate runs is not, so
|
||||||
|
# nobody has to read a repo to find out how to check it (D-27).
|
||||||
|
#
|
||||||
|
# Enable per clone with: git config core.hooksPath .githooks
|
||||||
|
# Never bypass with --no-verify. Suppress a specific finding deliberately
|
||||||
|
# instead, with a reason — see `make pre-push`.
|
||||||
|
set -euo pipefail
|
||||||
|
exec make -C "$(git rev-parse --show-toplevel)" pre-push
|
||||||
+28
-7
@@ -46,29 +46,37 @@ ENV/
|
|||||||
.ipynb_checkpoints/
|
.ipynb_checkpoints/
|
||||||
*.ipynb
|
*.ipynb
|
||||||
|
|
||||||
# Testing & Coverage
|
# Caches (pytest, mypy, ruff)
|
||||||
|
.cache/
|
||||||
|
|
||||||
|
# Build output (coverage, logs)
|
||||||
|
build/
|
||||||
|
|
||||||
|
# Legacy cache/output locations (in case tools fall back)
|
||||||
.pytest_cache/
|
.pytest_cache/
|
||||||
|
.mypy_cache/
|
||||||
|
.ruff_cache/
|
||||||
.coverage
|
.coverage
|
||||||
.coverage.*
|
|
||||||
coverage.xml
|
coverage.xml
|
||||||
htmlcov/
|
htmlcov/
|
||||||
|
|
||||||
|
# Testing
|
||||||
.tox/
|
.tox/
|
||||||
.nox/
|
.nox/
|
||||||
*.cover
|
*.cover
|
||||||
.hypothesis/
|
.hypothesis/
|
||||||
|
|
||||||
# Type checking
|
# Type checking
|
||||||
.mypy_cache/
|
|
||||||
.dmypy.json
|
.dmypy.json
|
||||||
dmypy.json
|
dmypy.json
|
||||||
.pyre/
|
.pyre/
|
||||||
.pytype/
|
.pytype/
|
||||||
|
|
||||||
# Linting
|
|
||||||
.ruff_cache/
|
|
||||||
|
|
||||||
# Logs
|
# Logs
|
||||||
logs/
|
logs/*
|
||||||
|
!logs/traces/
|
||||||
|
logs/traces/*
|
||||||
|
!logs/traces/viewer.html
|
||||||
*.log
|
*.log
|
||||||
|
|
||||||
# Database
|
# Database
|
||||||
@@ -95,3 +103,16 @@ ollama_data/
|
|||||||
ehthumbs.db
|
ehthumbs.db
|
||||||
Thumbs.db
|
Thumbs.db
|
||||||
Desktop.ini
|
Desktop.ini
|
||||||
|
|
||||||
|
# Claude Code user-specific settings
|
||||||
|
.claude/settings.local.json
|
||||||
|
.pql/*
|
||||||
|
!.pql/changelog/
|
||||||
|
|
||||||
|
# pql shims planted by `pql init` into the dir core.hooksPath points at.
|
||||||
|
# Per-clone: each embeds the absolute path of the pql binary that planted it.
|
||||||
|
# Only .githooks/pre-push is shared.
|
||||||
|
.githooks/pre-commit
|
||||||
|
.githooks/post-merge
|
||||||
|
.githooks/post-checkout
|
||||||
|
.githooks/post-rewrite
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
-- Changelog format marker, written by pql. Comments only: this file
|
||||||
|
-- is never executed — Import descends into the per-table directories
|
||||||
|
-- and does not read the changelog root.
|
||||||
|
--
|
||||||
|
-- A changelog carrying no marker is format 1, the shape that existed
|
||||||
|
-- before formats were versioned. An older format is migrated forward
|
||||||
|
-- by `pql plan upgrade` (and automatically from the post-merge hook);
|
||||||
|
-- a newer one is refused rather than replayed under rules this binary
|
||||||
|
-- does not know. See D-28 and docs/versions.md.
|
||||||
|
-- pql:changelog_format: 2.0.0
|
||||||
|
-- pql:written_by: 2.2.0
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
-- Auto-generated by pql init. CREATE TABLE statements
|
||||||
|
-- for the planning schema; per-table dir keeps the changelog
|
||||||
|
-- self-describing per D-15. CREATE TABLE IF NOT EXISTS is
|
||||||
|
-- idempotent so running schema files from each directory in
|
||||||
|
-- replay order is harmless.
|
||||||
|
--
|
||||||
|
-- Importer parses the markers below to detect schema drift
|
||||||
|
-- between the producing pql version and the local one — a
|
||||||
|
-- bumped canonical_version means projection rules changed
|
||||||
|
-- and replay must refuse rather than silently corrupt state.
|
||||||
|
-- pql:created_by: 2.2.0
|
||||||
|
-- pql:canonical_version: 2
|
||||||
|
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decisions (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('confirmed','question','rejected')),
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
status TEXT NOT NULL DEFAULT 'active'
|
||||||
|
CHECK(status IN ('active','superseded','resolved','open')),
|
||||||
|
date TEXT,
|
||||||
|
file_path TEXT NOT NULL,
|
||||||
|
synced_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decision_refs (
|
||||||
|
source_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
target_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
ref_type TEXT NOT NULL
|
||||||
|
CHECK(ref_type IN ('supersedes','references','resolves','depends_on','amends')),
|
||||||
|
note TEXT,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (source_id, target_id, ref_type)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Identity split (D-26): a ticket's stable, collision-proof identity is its
|
||||||
|
-- record_id (a locally-generated ULID, planning.NewRecordID); the friendly
|
||||||
|
-- T-NNN label lives in ticket_idmap and may be reconciled. Every structural
|
||||||
|
-- reference (parent, deps, history, labels) targets record_id, so a label
|
||||||
|
-- clash never corrupts the graph — only ticket_idmap needs a relabel.
|
||||||
|
CREATE TABLE IF NOT EXISTS tickets (
|
||||||
|
record_id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('initiative','epic','story','task','bug')),
|
||||||
|
parent_record_id TEXT REFERENCES tickets(record_id),
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
-- No CHECK enumeration: the ticket status vocabulary is per-vault
|
||||||
|
-- configurable (ticket_statuses in .pql/config.yaml). Validation lives
|
||||||
|
-- in Go (planning.StatusSet), so adding/renaming statuses needs no
|
||||||
|
-- schema change. The DEFAULT is a harmless fallback — CreateTicket
|
||||||
|
-- always inserts the configured default explicitly.
|
||||||
|
status TEXT NOT NULL DEFAULT 'backlog',
|
||||||
|
priority TEXT DEFAULT 'medium'
|
||||||
|
CHECK(priority IN ('critical','high','medium','low')),
|
||||||
|
assigned_to TEXT,
|
||||||
|
team TEXT,
|
||||||
|
decision_ref TEXT REFERENCES decisions(id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
-- ticket_idmap maps a record_id to its current friendly label (T-NNN).
|
||||||
|
-- ticket_id is intentionally NOT globally unique: two uncoordinated clones
|
||||||
|
-- can mint the same label, which surfaces as a duplicate-label collision
|
||||||
|
-- (detected at replay) and is fixed with "pql ticket relabel".
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_idmap (
|
||||||
|
record_id TEXT PRIMARY KEY REFERENCES tickets(record_id),
|
||||||
|
ticket_id TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_deps (
|
||||||
|
blocker_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
blocked_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (blocker_record_id, blocked_record_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_history (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
field TEXT NOT NULL,
|
||||||
|
old_value TEXT,
|
||||||
|
new_value TEXT,
|
||||||
|
changed_by TEXT,
|
||||||
|
changed_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT UNIQUE,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_labels (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
label TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (ticket_record_id, label)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL,
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_status ON tickets(status);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_team ON tickets(team);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_decision_ref ON tickets(decision_ref);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_assigned ON tickets(assigned_to);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_parent ON tickets(parent_record_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_ticket_idmap_label ON ticket_idmap(ticket_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_domain ON decisions(domain);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_type ON decisions(type);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decision_refs_target ON decision_refs(target_id);
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
-- Auto-generated by pql init. CREATE TABLE statements
|
||||||
|
-- for the planning schema; per-table dir keeps the changelog
|
||||||
|
-- self-describing per D-15. CREATE TABLE IF NOT EXISTS is
|
||||||
|
-- idempotent so running schema files from each directory in
|
||||||
|
-- replay order is harmless.
|
||||||
|
--
|
||||||
|
-- Importer parses the markers below to detect schema drift
|
||||||
|
-- between the producing pql version and the local one — a
|
||||||
|
-- bumped canonical_version means projection rules changed
|
||||||
|
-- and replay must refuse rather than silently corrupt state.
|
||||||
|
-- pql:created_by: 2.2.0
|
||||||
|
-- pql:canonical_version: 2
|
||||||
|
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decisions (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('confirmed','question','rejected')),
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
status TEXT NOT NULL DEFAULT 'active'
|
||||||
|
CHECK(status IN ('active','superseded','resolved','open')),
|
||||||
|
date TEXT,
|
||||||
|
file_path TEXT NOT NULL,
|
||||||
|
synced_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decision_refs (
|
||||||
|
source_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
target_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
ref_type TEXT NOT NULL
|
||||||
|
CHECK(ref_type IN ('supersedes','references','resolves','depends_on','amends')),
|
||||||
|
note TEXT,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (source_id, target_id, ref_type)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Identity split (D-26): a ticket's stable, collision-proof identity is its
|
||||||
|
-- record_id (a locally-generated ULID, planning.NewRecordID); the friendly
|
||||||
|
-- T-NNN label lives in ticket_idmap and may be reconciled. Every structural
|
||||||
|
-- reference (parent, deps, history, labels) targets record_id, so a label
|
||||||
|
-- clash never corrupts the graph — only ticket_idmap needs a relabel.
|
||||||
|
CREATE TABLE IF NOT EXISTS tickets (
|
||||||
|
record_id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('initiative','epic','story','task','bug')),
|
||||||
|
parent_record_id TEXT REFERENCES tickets(record_id),
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
-- No CHECK enumeration: the ticket status vocabulary is per-vault
|
||||||
|
-- configurable (ticket_statuses in .pql/config.yaml). Validation lives
|
||||||
|
-- in Go (planning.StatusSet), so adding/renaming statuses needs no
|
||||||
|
-- schema change. The DEFAULT is a harmless fallback — CreateTicket
|
||||||
|
-- always inserts the configured default explicitly.
|
||||||
|
status TEXT NOT NULL DEFAULT 'backlog',
|
||||||
|
priority TEXT DEFAULT 'medium'
|
||||||
|
CHECK(priority IN ('critical','high','medium','low')),
|
||||||
|
assigned_to TEXT,
|
||||||
|
team TEXT,
|
||||||
|
decision_ref TEXT REFERENCES decisions(id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
-- ticket_idmap maps a record_id to its current friendly label (T-NNN).
|
||||||
|
-- ticket_id is intentionally NOT globally unique: two uncoordinated clones
|
||||||
|
-- can mint the same label, which surfaces as a duplicate-label collision
|
||||||
|
-- (detected at replay) and is fixed with "pql ticket relabel".
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_idmap (
|
||||||
|
record_id TEXT PRIMARY KEY REFERENCES tickets(record_id),
|
||||||
|
ticket_id TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_deps (
|
||||||
|
blocker_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
blocked_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (blocker_record_id, blocked_record_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_history (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
field TEXT NOT NULL,
|
||||||
|
old_value TEXT,
|
||||||
|
new_value TEXT,
|
||||||
|
changed_by TEXT,
|
||||||
|
changed_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT UNIQUE,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_labels (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
label TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (ticket_record_id, label)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL,
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_status ON tickets(status);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_team ON tickets(team);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_decision_ref ON tickets(decision_ref);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_assigned ON tickets(assigned_to);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_parent ON tickets(parent_record_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_ticket_idmap_label ON ticket_idmap(ticket_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_domain ON decisions(domain);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_type ON decisions(type);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decision_refs_target ON decision_refs(target_id);
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
INSERT INTO ticket_history (ticket_record_id, field, old_value, new_value, changed_by, changed_at, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ44Z6HSN0RQS0QAYNTPEM5G', 'description', NULL, '`make typecheck` reports 95 errors in 31 files (was 103). This is a dedicated programming pass, not lint tidying, and it is what currently blocks `make pre-push`.
|
||||||
|
|
||||||
|
MEASURED 2026-08-11 so the next session does not re-derive it:
|
||||||
|
|
||||||
|
29 no-untyped-def functions with no annotations — the bulk, and genuine per-function work
|
||||||
|
18 no-any-return mostly downstream of the above
|
||||||
|
12 assignment
|
||||||
|
11 arg-type
|
||||||
|
5 unused-ignore `# type: ignore` comments mypy says are no longer needed
|
||||||
|
5 union-attr
|
||||||
|
4 var-annotated
|
||||||
|
3 override
|
||||||
|
remainder: misc, dict-item, attr-defined, return-value, call-overload
|
||||||
|
|
||||||
|
By file: agents/tatlock.py 15, core/memory_service.py 11, responses/streaming.py 8, responses/service.py 7, core/context.py 6.
|
||||||
|
|
||||||
|
THE TWO SHARED ROOTS ARE ALREADY FIXED (dac259a), so what is left has no lever in it. For reference, they were: five conversation lists declared bare, where mypy infers the element type from the first append (a ModelRequest) and then rejects every ModelResponse; and an agent built as Agent(model, system_prompt=...) with no deps_type, inferred Agent[None, str], while every tool it registers takes RunContext[ToolCallTracker].
|
||||||
|
|
||||||
|
WORTH KNOWING BEFORE STARTING. Annotating partially made mypy count go UP before it went down — declaring `_agent: Agent | None` took agents/tatlock.py from 22 to 24, because resolving the bare Agent to Agent[None, str] surfaced four argument-type errors the Any had been hiding. Expect that shape: a rising count during this work usually means concealment ending, not damage.
|
||||||
|
|
||||||
|
The mypy config is strict — disallow_untyped_defs, disallow_incomplete_defs, warn_return_any, check_untyped_defs, strict_equality — so there is no partial-credit setting to lean on, and weakening it would be the wrong trade for a codebase this central.
|
||||||
|
|
||||||
|
TWO PRE-EXISTING TEST FACTS, both confirmed at HEAD and neither caused by the lint work:
|
||||||
|
- test_tatlock_tool_call_logging_calculator is flaky: failed 2 of 5 full runs, on HEAD and on the lint branch, and fails in isolation at HEAD while passing in isolation after the lint pass. Order- or timing-dependent.
|
||||||
|
- `pytest tests/` cannot collect at all: tests/e2e/test_orchestration_e2e.py uses an `e2e` marker that is not registered and the config is strict about markers. `make test` passes only because it ignores tests/e2e, tests/integration and tests/contracts.', NULL, '2026-08-11 18:26:58', '2026-08-11 18:26:58.523', '2026-08-11 18:26:58.523', NULL, 'c8c9b6a20cf18ca903fc8c720f70e73d', 2) ON CONFLICT(hash) DO NOTHING;
|
||||||
|
INSERT INTO ticket_history (ticket_record_id, field, old_value, new_value, changed_by, changed_at, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ4JX9DZB3XAB95EP23YWXMC', 'description', NULL, '`.env` still carries an `ANTHROPIC_API_KEY`. That key was revoked and no longer exists on Anthropic''s side, so this is dead weight rather than an exposure — but it is dead weight that reads exactly like a live credential to anyone who finds it.
|
||||||
|
|
||||||
|
The cost is confusion, not risk. Someone debugging a Claude fallback will find a key present, assume it is configured, and look elsewhere for the failure. The absence of a key is a clear signal; a revoked key is a misleading one.
|
||||||
|
|
||||||
|
`.env` is gitignored here and has never been committed, so nothing needs rewriting — the value simply needs removing from the local file, and the line dropping or blanking in `.env.example` if it appears there too.
|
||||||
|
|
||||||
|
RELATED, and the reason this is filed separately: the workspace vault holds T-13, "Clear the revoked ANTHROPIC_API_KEY from the live Portainer stack". That ticket is scoped to the Portainer stack only. Whoever closes it will reasonably believe the key is gone once the stack is clean, and this copy will survive. The two want doing together even though they commit separately.
|
||||||
|
|
||||||
|
Context on the revocation, since it explains why nobody removed this at the time: the key was revoked on 2026-08-09 after being printed into a transcript by a redaction filter that matched on `KEY` appearing after the `=`. In `ANTHROPIC_API_KEY=...` it appears before, so the filter never fired. The response was rotation, and the leftover copies were not swept.', NULL, '2026-08-11 19:27:52', '2026-08-11 19:27:52.823', '2026-08-11 19:27:52.823', NULL, 'c7834e46268029b74c25a25a83177b64', 2) ON CONFLICT(hash) DO NOTHING;
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
-- Auto-generated by pql init. CREATE TABLE statements
|
||||||
|
-- for the planning schema; per-table dir keeps the changelog
|
||||||
|
-- self-describing per D-15. CREATE TABLE IF NOT EXISTS is
|
||||||
|
-- idempotent so running schema files from each directory in
|
||||||
|
-- replay order is harmless.
|
||||||
|
--
|
||||||
|
-- Importer parses the markers below to detect schema drift
|
||||||
|
-- between the producing pql version and the local one — a
|
||||||
|
-- bumped canonical_version means projection rules changed
|
||||||
|
-- and replay must refuse rather than silently corrupt state.
|
||||||
|
-- pql:created_by: 2.2.0
|
||||||
|
-- pql:canonical_version: 2
|
||||||
|
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decisions (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('confirmed','question','rejected')),
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
status TEXT NOT NULL DEFAULT 'active'
|
||||||
|
CHECK(status IN ('active','superseded','resolved','open')),
|
||||||
|
date TEXT,
|
||||||
|
file_path TEXT NOT NULL,
|
||||||
|
synced_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decision_refs (
|
||||||
|
source_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
target_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
ref_type TEXT NOT NULL
|
||||||
|
CHECK(ref_type IN ('supersedes','references','resolves','depends_on','amends')),
|
||||||
|
note TEXT,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (source_id, target_id, ref_type)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Identity split (D-26): a ticket's stable, collision-proof identity is its
|
||||||
|
-- record_id (a locally-generated ULID, planning.NewRecordID); the friendly
|
||||||
|
-- T-NNN label lives in ticket_idmap and may be reconciled. Every structural
|
||||||
|
-- reference (parent, deps, history, labels) targets record_id, so a label
|
||||||
|
-- clash never corrupts the graph — only ticket_idmap needs a relabel.
|
||||||
|
CREATE TABLE IF NOT EXISTS tickets (
|
||||||
|
record_id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('initiative','epic','story','task','bug')),
|
||||||
|
parent_record_id TEXT REFERENCES tickets(record_id),
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
-- No CHECK enumeration: the ticket status vocabulary is per-vault
|
||||||
|
-- configurable (ticket_statuses in .pql/config.yaml). Validation lives
|
||||||
|
-- in Go (planning.StatusSet), so adding/renaming statuses needs no
|
||||||
|
-- schema change. The DEFAULT is a harmless fallback — CreateTicket
|
||||||
|
-- always inserts the configured default explicitly.
|
||||||
|
status TEXT NOT NULL DEFAULT 'backlog',
|
||||||
|
priority TEXT DEFAULT 'medium'
|
||||||
|
CHECK(priority IN ('critical','high','medium','low')),
|
||||||
|
assigned_to TEXT,
|
||||||
|
team TEXT,
|
||||||
|
decision_ref TEXT REFERENCES decisions(id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
-- ticket_idmap maps a record_id to its current friendly label (T-NNN).
|
||||||
|
-- ticket_id is intentionally NOT globally unique: two uncoordinated clones
|
||||||
|
-- can mint the same label, which surfaces as a duplicate-label collision
|
||||||
|
-- (detected at replay) and is fixed with "pql ticket relabel".
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_idmap (
|
||||||
|
record_id TEXT PRIMARY KEY REFERENCES tickets(record_id),
|
||||||
|
ticket_id TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_deps (
|
||||||
|
blocker_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
blocked_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (blocker_record_id, blocked_record_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_history (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
field TEXT NOT NULL,
|
||||||
|
old_value TEXT,
|
||||||
|
new_value TEXT,
|
||||||
|
changed_by TEXT,
|
||||||
|
changed_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT UNIQUE,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_labels (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
label TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (ticket_record_id, label)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL,
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_status ON tickets(status);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_team ON tickets(team);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_decision_ref ON tickets(decision_ref);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_assigned ON tickets(assigned_to);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_parent ON tickets(parent_record_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_ticket_idmap_label ON ticket_idmap(ticket_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_domain ON decisions(domain);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_type ON decisions(type);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decision_refs_target ON decision_refs(target_id);
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
INSERT INTO ticket_idmap (record_id, ticket_id, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ44Z6HSN0RQS0QAYNTPEM5G', 'T-1', '2026-08-11 18:26:58.365', '2026-08-11 18:26:58.365', NULL, 'f1508986553f1ee59145a0d099131a68', 2) ON CONFLICT(record_id) DO UPDATE SET ticket_id=excluded.ticket_id, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= ticket_idmap.updated_at;
|
||||||
|
INSERT INTO ticket_idmap (record_id, ticket_id, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ4JX9DZB3XAB95EP23YWXMC', 'T-2', '2026-08-11 19:27:52.688', '2026-08-11 19:27:52.688', NULL, 'a5b18acb4b9a7e2da8e47b6ee603a5da', 2) ON CONFLICT(record_id) DO UPDATE SET ticket_id=excluded.ticket_id, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= ticket_idmap.updated_at;
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
-- Auto-generated by pql init. CREATE TABLE statements
|
||||||
|
-- for the planning schema; per-table dir keeps the changelog
|
||||||
|
-- self-describing per D-15. CREATE TABLE IF NOT EXISTS is
|
||||||
|
-- idempotent so running schema files from each directory in
|
||||||
|
-- replay order is harmless.
|
||||||
|
--
|
||||||
|
-- Importer parses the markers below to detect schema drift
|
||||||
|
-- between the producing pql version and the local one — a
|
||||||
|
-- bumped canonical_version means projection rules changed
|
||||||
|
-- and replay must refuse rather than silently corrupt state.
|
||||||
|
-- pql:created_by: 2.2.0
|
||||||
|
-- pql:canonical_version: 2
|
||||||
|
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decisions (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('confirmed','question','rejected')),
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
status TEXT NOT NULL DEFAULT 'active'
|
||||||
|
CHECK(status IN ('active','superseded','resolved','open')),
|
||||||
|
date TEXT,
|
||||||
|
file_path TEXT NOT NULL,
|
||||||
|
synced_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decision_refs (
|
||||||
|
source_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
target_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
ref_type TEXT NOT NULL
|
||||||
|
CHECK(ref_type IN ('supersedes','references','resolves','depends_on','amends')),
|
||||||
|
note TEXT,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (source_id, target_id, ref_type)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Identity split (D-26): a ticket's stable, collision-proof identity is its
|
||||||
|
-- record_id (a locally-generated ULID, planning.NewRecordID); the friendly
|
||||||
|
-- T-NNN label lives in ticket_idmap and may be reconciled. Every structural
|
||||||
|
-- reference (parent, deps, history, labels) targets record_id, so a label
|
||||||
|
-- clash never corrupts the graph — only ticket_idmap needs a relabel.
|
||||||
|
CREATE TABLE IF NOT EXISTS tickets (
|
||||||
|
record_id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('initiative','epic','story','task','bug')),
|
||||||
|
parent_record_id TEXT REFERENCES tickets(record_id),
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
-- No CHECK enumeration: the ticket status vocabulary is per-vault
|
||||||
|
-- configurable (ticket_statuses in .pql/config.yaml). Validation lives
|
||||||
|
-- in Go (planning.StatusSet), so adding/renaming statuses needs no
|
||||||
|
-- schema change. The DEFAULT is a harmless fallback — CreateTicket
|
||||||
|
-- always inserts the configured default explicitly.
|
||||||
|
status TEXT NOT NULL DEFAULT 'backlog',
|
||||||
|
priority TEXT DEFAULT 'medium'
|
||||||
|
CHECK(priority IN ('critical','high','medium','low')),
|
||||||
|
assigned_to TEXT,
|
||||||
|
team TEXT,
|
||||||
|
decision_ref TEXT REFERENCES decisions(id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
-- ticket_idmap maps a record_id to its current friendly label (T-NNN).
|
||||||
|
-- ticket_id is intentionally NOT globally unique: two uncoordinated clones
|
||||||
|
-- can mint the same label, which surfaces as a duplicate-label collision
|
||||||
|
-- (detected at replay) and is fixed with "pql ticket relabel".
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_idmap (
|
||||||
|
record_id TEXT PRIMARY KEY REFERENCES tickets(record_id),
|
||||||
|
ticket_id TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_deps (
|
||||||
|
blocker_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
blocked_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (blocker_record_id, blocked_record_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_history (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
field TEXT NOT NULL,
|
||||||
|
old_value TEXT,
|
||||||
|
new_value TEXT,
|
||||||
|
changed_by TEXT,
|
||||||
|
changed_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT UNIQUE,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_labels (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
label TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (ticket_record_id, label)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL,
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_status ON tickets(status);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_team ON tickets(team);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_decision_ref ON tickets(decision_ref);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_assigned ON tickets(assigned_to);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_parent ON tickets(parent_record_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_ticket_idmap_label ON ticket_idmap(ticket_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_domain ON decisions(domain);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_type ON decisions(type);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decision_refs_target ON decision_refs(target_id);
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
-- Auto-generated by pql init. CREATE TABLE statements
|
||||||
|
-- for the planning schema; per-table dir keeps the changelog
|
||||||
|
-- self-describing per D-15. CREATE TABLE IF NOT EXISTS is
|
||||||
|
-- idempotent so running schema files from each directory in
|
||||||
|
-- replay order is harmless.
|
||||||
|
--
|
||||||
|
-- Importer parses the markers below to detect schema drift
|
||||||
|
-- between the producing pql version and the local one — a
|
||||||
|
-- bumped canonical_version means projection rules changed
|
||||||
|
-- and replay must refuse rather than silently corrupt state.
|
||||||
|
-- pql:created_by: 2.2.0
|
||||||
|
-- pql:canonical_version: 2
|
||||||
|
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decisions (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('confirmed','question','rejected')),
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
status TEXT NOT NULL DEFAULT 'active'
|
||||||
|
CHECK(status IN ('active','superseded','resolved','open')),
|
||||||
|
date TEXT,
|
||||||
|
file_path TEXT NOT NULL,
|
||||||
|
synced_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS decision_refs (
|
||||||
|
source_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
target_id TEXT NOT NULL REFERENCES decisions(id) ON DELETE CASCADE,
|
||||||
|
ref_type TEXT NOT NULL
|
||||||
|
CHECK(ref_type IN ('supersedes','references','resolves','depends_on','amends')),
|
||||||
|
note TEXT,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (source_id, target_id, ref_type)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Identity split (D-26): a ticket's stable, collision-proof identity is its
|
||||||
|
-- record_id (a locally-generated ULID, planning.NewRecordID); the friendly
|
||||||
|
-- T-NNN label lives in ticket_idmap and may be reconciled. Every structural
|
||||||
|
-- reference (parent, deps, history, labels) targets record_id, so a label
|
||||||
|
-- clash never corrupts the graph — only ticket_idmap needs a relabel.
|
||||||
|
CREATE TABLE IF NOT EXISTS tickets (
|
||||||
|
record_id TEXT PRIMARY KEY,
|
||||||
|
type TEXT NOT NULL CHECK(type IN ('initiative','epic','story','task','bug')),
|
||||||
|
parent_record_id TEXT REFERENCES tickets(record_id),
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
-- No CHECK enumeration: the ticket status vocabulary is per-vault
|
||||||
|
-- configurable (ticket_statuses in .pql/config.yaml). Validation lives
|
||||||
|
-- in Go (planning.StatusSet), so adding/renaming statuses needs no
|
||||||
|
-- schema change. The DEFAULT is a harmless fallback — CreateTicket
|
||||||
|
-- always inserts the configured default explicitly.
|
||||||
|
status TEXT NOT NULL DEFAULT 'backlog',
|
||||||
|
priority TEXT DEFAULT 'medium'
|
||||||
|
CHECK(priority IN ('critical','high','medium','low')),
|
||||||
|
assigned_to TEXT,
|
||||||
|
team TEXT,
|
||||||
|
decision_ref TEXT REFERENCES decisions(id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
-- ticket_idmap maps a record_id to its current friendly label (T-NNN).
|
||||||
|
-- ticket_id is intentionally NOT globally unique: two uncoordinated clones
|
||||||
|
-- can mint the same label, which surfaces as a duplicate-label collision
|
||||||
|
-- (detected at replay) and is fixed with "pql ticket relabel".
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_idmap (
|
||||||
|
record_id TEXT PRIMARY KEY REFERENCES tickets(record_id),
|
||||||
|
ticket_id TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_deps (
|
||||||
|
blocker_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
blocked_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (blocker_record_id, blocked_record_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_history (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
field TEXT NOT NULL,
|
||||||
|
old_value TEXT,
|
||||||
|
new_value TEXT,
|
||||||
|
changed_by TEXT,
|
||||||
|
changed_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT UNIQUE,
|
||||||
|
canonical_version INTEGER
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS ticket_labels (
|
||||||
|
ticket_record_id TEXT NOT NULL REFERENCES tickets(record_id),
|
||||||
|
label TEXT NOT NULL,
|
||||||
|
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||||
|
deleted_at TEXT,
|
||||||
|
hash TEXT,
|
||||||
|
canonical_version INTEGER,
|
||||||
|
PRIMARY KEY (ticket_record_id, label)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL,
|
||||||
|
updated_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_status ON tickets(status);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_team ON tickets(team);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_decision_ref ON tickets(decision_ref);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_assigned ON tickets(assigned_to);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_tickets_parent ON tickets(parent_record_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_ticket_idmap_label ON ticket_idmap(ticket_id);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_domain ON decisions(domain);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decisions_type ON decisions(type);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_decision_refs_target ON decision_refs(target_id);
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ44Z6HSN0RQS0QAYNTPEM5G', 'task', NULL, 'Type the codebase: 95 mypy errors across 31 files', NULL, 'backlog', 'high', NULL, NULL, NULL, '2026-08-11 18:26:58.318', '2026-08-11 18:26:58.318', NULL, 'd881f36d77e58aaa2f94dc08c5b5be3e', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at;
|
||||||
|
INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ44Z6HSN0RQS0QAYNTPEM5G', 'task', NULL, 'Type the codebase: 95 mypy errors across 31 files', '`make typecheck` reports 95 errors in 31 files (was 103). This is a dedicated programming pass, not lint tidying, and it is what currently blocks `make pre-push`.
|
||||||
|
|
||||||
|
MEASURED 2026-08-11 so the next session does not re-derive it:
|
||||||
|
|
||||||
|
29 no-untyped-def functions with no annotations — the bulk, and genuine per-function work
|
||||||
|
18 no-any-return mostly downstream of the above
|
||||||
|
12 assignment
|
||||||
|
11 arg-type
|
||||||
|
5 unused-ignore `# type: ignore` comments mypy says are no longer needed
|
||||||
|
5 union-attr
|
||||||
|
4 var-annotated
|
||||||
|
3 override
|
||||||
|
remainder: misc, dict-item, attr-defined, return-value, call-overload
|
||||||
|
|
||||||
|
By file: agents/tatlock.py 15, core/memory_service.py 11, responses/streaming.py 8, responses/service.py 7, core/context.py 6.
|
||||||
|
|
||||||
|
THE TWO SHARED ROOTS ARE ALREADY FIXED (dac259a), so what is left has no lever in it. For reference, they were: five conversation lists declared bare, where mypy infers the element type from the first append (a ModelRequest) and then rejects every ModelResponse; and an agent built as Agent(model, system_prompt=...) with no deps_type, inferred Agent[None, str], while every tool it registers takes RunContext[ToolCallTracker].
|
||||||
|
|
||||||
|
WORTH KNOWING BEFORE STARTING. Annotating partially made mypy count go UP before it went down — declaring `_agent: Agent | None` took agents/tatlock.py from 22 to 24, because resolving the bare Agent to Agent[None, str] surfaced four argument-type errors the Any had been hiding. Expect that shape: a rising count during this work usually means concealment ending, not damage.
|
||||||
|
|
||||||
|
The mypy config is strict — disallow_untyped_defs, disallow_incomplete_defs, warn_return_any, check_untyped_defs, strict_equality — so there is no partial-credit setting to lean on, and weakening it would be the wrong trade for a codebase this central.
|
||||||
|
|
||||||
|
TWO PRE-EXISTING TEST FACTS, both confirmed at HEAD and neither caused by the lint work:
|
||||||
|
- test_tatlock_tool_call_logging_calculator is flaky: failed 2 of 5 full runs, on HEAD and on the lint branch, and fails in isolation at HEAD while passing in isolation after the lint pass. Order- or timing-dependent.
|
||||||
|
- `pytest tests/` cannot collect at all: tests/e2e/test_orchestration_e2e.py uses an `e2e` marker that is not registered and the config is strict about markers. `make test` passes only because it ignores tests/e2e, tests/integration and tests/contracts.', 'backlog', 'high', NULL, NULL, NULL, '2026-08-11 18:26:58.318', '2026-08-11 18:26:58.523', NULL, 'fc047dd080d976c4ca0c551cac8fec93', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at;
|
||||||
|
INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ4JX9DZB3XAB95EP23YWXMC', 'bug', NULL, 'A revoked ANTHROPIC_API_KEY is still sitting in .env', NULL, 'backlog', 'medium', NULL, NULL, NULL, '2026-08-11 19:27:52.687', '2026-08-11 19:27:52.687', NULL, '3b00a4b729819e20a6425f971e6cb9da', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at;
|
||||||
|
INSERT INTO tickets (record_id, type, parent_record_id, title, description, status, priority, assigned_to, team, decision_ref, created_at, updated_at, deleted_at, hash, canonical_version) VALUES ('06FZ4JX9DZB3XAB95EP23YWXMC', 'bug', NULL, 'A revoked ANTHROPIC_API_KEY is still sitting in .env', '`.env` still carries an `ANTHROPIC_API_KEY`. That key was revoked and no longer exists on Anthropic''s side, so this is dead weight rather than an exposure — but it is dead weight that reads exactly like a live credential to anyone who finds it.
|
||||||
|
|
||||||
|
The cost is confusion, not risk. Someone debugging a Claude fallback will find a key present, assume it is configured, and look elsewhere for the failure. The absence of a key is a clear signal; a revoked key is a misleading one.
|
||||||
|
|
||||||
|
`.env` is gitignored here and has never been committed, so nothing needs rewriting — the value simply needs removing from the local file, and the line dropping or blanking in `.env.example` if it appears there too.
|
||||||
|
|
||||||
|
RELATED, and the reason this is filed separately: the workspace vault holds T-13, "Clear the revoked ANTHROPIC_API_KEY from the live Portainer stack". That ticket is scoped to the Portainer stack only. Whoever closes it will reasonably believe the key is gone once the stack is clean, and this copy will survive. The two want doing together even though they commit separately.
|
||||||
|
|
||||||
|
Context on the revocation, since it explains why nobody removed this at the time: the key was revoked on 2026-08-09 after being printed into a transcript by a redaction filter that matched on `KEY` appearing after the `=`. In `ANTHROPIC_API_KEY=...` it appears before, so the filter never fired. The response was rotation, and the leftover copies were not swept.', 'backlog', 'medium', NULL, NULL, NULL, '2026-08-11 19:27:52.687', '2026-08-11 19:27:52.823', NULL, 'a9c2ca965b3d40a924720c7761c5338d', 2) ON CONFLICT(record_id) DO UPDATE SET type=excluded.type, parent_record_id=excluded.parent_record_id, title=excluded.title, description=excluded.description, status=excluded.status, priority=excluded.priority, assigned_to=excluded.assigned_to, team=excluded.team, decision_ref=excluded.decision_ref, updated_at=excluded.updated_at, deleted_at=excluded.deleted_at, hash=excluded.hash, canonical_version=excluded.canonical_version WHERE excluded.updated_at >= tickets.updated_at;
|
||||||
@@ -1,544 +0,0 @@
|
|||||||
# LLM Agent Instructions
|
|
||||||
|
|
||||||
This document contains instructions and documentation references for AI assistants working with this codebase.
|
|
||||||
|
|
||||||
> **📖 Important**: Before working on this project, read [PHILOSOPHY.md](PHILOSOPHY.md) to understand the system vision, architectural patterns, and design goals. All development should work towards realizing those patterns.
|
|
||||||
|
|
||||||
## Project Overview
|
|
||||||
|
|
||||||
This project implements an OpenAI-compatible API with FastAPI, featuring a hybrid architecture that provides both the OpenAI Responses API and Chat Completions compatibility layer.
|
|
||||||
|
|
||||||
### Architecture Pattern
|
|
||||||
|
|
||||||
The **Orchestrator** infrastructure layer with hybrid API architecture:
|
|
||||||
|
|
||||||
```
|
|
||||||
Client (Open WebUI)
|
|
||||||
↓
|
|
||||||
Chat Completions (/v1/chat/completions) → Wrapper
|
|
||||||
↓
|
|
||||||
Responses API (/v1/responses) → Primary
|
|
||||||
↓
|
|
||||||
Agent Interface (lorem-tester, Tatlock)
|
|
||||||
↓
|
|
||||||
Mock Agents (lorem-tester) / Future: PydanticAI Agents (Tatlock, Steward, etc.)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Architectural Layers:**
|
|
||||||
|
|
||||||
1. **The Orchestrator** (Current Implementation)
|
|
||||||
- FastAPI application providing the infrastructure
|
|
||||||
- HTTP/SSE endpoints, streaming coordination
|
|
||||||
- Conversation history and context management
|
|
||||||
- OpenAI-compatible API surface
|
|
||||||
|
|
||||||
2. **Future: The Household** (Phases 1-4)
|
|
||||||
- **Steward**: First-tier LLM for request analysis (PydanticAI agent)
|
|
||||||
- **Tatlock**: Second-tier LLM with butler personality (PydanticAI agent)
|
|
||||||
- **Expert Agents**: Domain specialists (Librarian, Developer, Handyman, etc.)
|
|
||||||
|
|
||||||
**Key Architectural Decisions:**
|
|
||||||
|
|
||||||
1. **Single Source of Truth**: All response generation happens in the Responses API
|
|
||||||
- Structured output with reasoning, function_call, and message items
|
|
||||||
- Real-time stop sequence and max tokens enforcement
|
|
||||||
- Conversation history tracking
|
|
||||||
- Context window management
|
|
||||||
|
|
||||||
2. **Chat Completions Wrapper**: Provides compatibility without duplicating logic
|
|
||||||
- Calls Responses API internally
|
|
||||||
- Automatically enables reasoning generation
|
|
||||||
- Converts reasoning items to `<think>` tags for Open WebUI
|
|
||||||
- Maintains OpenAI-compatible format
|
|
||||||
|
|
||||||
3. **Agent Interface**: Clean abstraction for multiple models
|
|
||||||
- **lorem-tester**: Full-featured mock agent with realistic behavior
|
|
||||||
- Reasoning summaries (adjustable effort levels)
|
|
||||||
- Random tool/function calls
|
|
||||||
- Error triggers for testing
|
|
||||||
- Temperature variation
|
|
||||||
- **Tatlock**: Advertised model name (currently mock, future: PydanticAI Butler agent)
|
|
||||||
|
|
||||||
4. **Hybrid Conversation History**:
|
|
||||||
- Client MUST send full context in `input` array (OpenAI compatible)
|
|
||||||
- Server optionally tracks via `metadata.conversation_id`
|
|
||||||
- Auto-generates deterministic IDs from first message
|
|
||||||
- Supports future vector memory integration (Qdrant)
|
|
||||||
|
|
||||||
**Why This Architecture?**
|
|
||||||
|
|
||||||
- **Open WebUI Compatibility**: Native Responses API support not yet in stable release
|
|
||||||
- **Future-Proof**: Easy migration when Open WebUI adds native support
|
|
||||||
- **Testability**: Full-featured mock agent (lorem-tester) for integration testing
|
|
||||||
- **Clean Separation**: Responses API as stable core, wrappers can change
|
|
||||||
|
|
||||||
### Components
|
|
||||||
|
|
||||||
- **FastAPI**: Web framework for the API layer
|
|
||||||
- **SSE-Starlette**: Server-Sent Events for streaming responses
|
|
||||||
- **Pydantic**: Request/response validation with field validators
|
|
||||||
- **Agent Interface**: Abstract base class for model implementations
|
|
||||||
- **Conversation History**: Server-side tracking with configurable max turns
|
|
||||||
- **Context Window**: Token counting and management
|
|
||||||
- **PydanticAI**: Integrated with Tatlock agent (Ollama backend)
|
|
||||||
- **Agent Tools**: Permanent tools module (`src/agents/tools.py`)
|
|
||||||
- Calculator: Safe mathematical expression evaluation
|
|
||||||
- Date/Time toolkit: Current time, relative dates, time differences
|
|
||||||
- Web Search: SearXNG integration for privacy-preserving search
|
|
||||||
|
|
||||||
## Documentation References
|
|
||||||
|
|
||||||
### Core Framework Documentation
|
|
||||||
|
|
||||||
#### FastAPI
|
|
||||||
- **Official Documentation**: https://fastapi.tiangolo.com/
|
|
||||||
- **Version**: 0.123.9 (Dec 2025)
|
|
||||||
- **Key Topics**:
|
|
||||||
- Path operations and routing
|
|
||||||
- Request/response models with Pydantic
|
|
||||||
- Dependency injection
|
|
||||||
- Background tasks
|
|
||||||
- WebSocket and streaming support
|
|
||||||
- **PyPI**: https://pypi.org/project/fastapi/
|
|
||||||
|
|
||||||
#### Uvicorn
|
|
||||||
- **Official Documentation**: https://www.uvicorn.org/
|
|
||||||
- **Version**: 0.38.0 (Oct 2025)
|
|
||||||
- **Key Topics**:
|
|
||||||
- ASGI server configuration
|
|
||||||
- Deployment settings
|
|
||||||
- Logging and monitoring
|
|
||||||
- SSL/TLS configuration
|
|
||||||
|
|
||||||
### AI/LLM Integration
|
|
||||||
|
|
||||||
#### PydanticAI
|
|
||||||
- **Official Documentation**: https://ai.pydantic.dev/
|
|
||||||
- **Version**: 1.27.0 (Dec 2025)
|
|
||||||
- **Status**: Dependency installed, ready for future integration
|
|
||||||
- **Key Topics** (for future implementation):
|
|
||||||
- Agent creation and configuration
|
|
||||||
- LLM provider integration (Ollama support)
|
|
||||||
- Structured outputs with Pydantic
|
|
||||||
- Streaming responses
|
|
||||||
- Tool/function calling
|
|
||||||
- RunContext and dynamic configuration
|
|
||||||
- MCP server integration
|
|
||||||
- **GitHub**: https://github.com/pydantic/pydantic-ai
|
|
||||||
- **PyPI**: https://pypi.org/project/pydantic-ai/
|
|
||||||
|
|
||||||
#### Pydantic
|
|
||||||
- **Official Documentation**: https://docs.pydantic.dev/latest/
|
|
||||||
- **Version**: 2.11+ (Required for PydanticAI, currently using >=2.11,<2.13)
|
|
||||||
- **Key Topics**:
|
|
||||||
- Data validation and serialization
|
|
||||||
- Field types and validators
|
|
||||||
- Model configuration
|
|
||||||
- JSON schema generation
|
|
||||||
|
|
||||||
### HTTP and Streaming
|
|
||||||
|
|
||||||
#### HTTPX
|
|
||||||
- **Official Documentation**: https://www.python-httpx.org/
|
|
||||||
- **Version**: 0.28.1
|
|
||||||
- **Key Topics**:
|
|
||||||
- Async HTTP client for Ollama communication
|
|
||||||
- Streaming responses
|
|
||||||
- Timeout configuration
|
|
||||||
- Connection pooling
|
|
||||||
|
|
||||||
#### SSE-Starlette
|
|
||||||
- **GitHub**: https://github.com/sysid/sse-starlette
|
|
||||||
- **Version**: 3.0.2 (Oct 2025)
|
|
||||||
- **Key Topics**:
|
|
||||||
- Server-Sent Events implementation
|
|
||||||
- Streaming event responses
|
|
||||||
- Integration with FastAPI/Starlette
|
|
||||||
|
|
||||||
### Ollama Integration
|
|
||||||
|
|
||||||
#### Ollama API
|
|
||||||
- **Official Documentation**: https://github.com/ollama/ollama/blob/main/docs/api.md
|
|
||||||
- **Status**: Async client implemented in `src/ollama/client.py`, ready for future integration
|
|
||||||
- **Key Topics** (for future implementation):
|
|
||||||
- REST API endpoints
|
|
||||||
- Streaming responses
|
|
||||||
- Model management
|
|
||||||
- Generate and chat endpoints
|
|
||||||
- Model configuration
|
|
||||||
- **Current Model Target**: mistral-nemo:latest
|
|
||||||
|
|
||||||
### OpenAI API Compatibility
|
|
||||||
|
|
||||||
#### OpenAI API Reference
|
|
||||||
- **Official Documentation**: https://platform.openai.com/docs/api-reference
|
|
||||||
- **Key API Endpoints**:
|
|
||||||
- `/v1/responses` - Responses API (PRIMARY) with structured output
|
|
||||||
- `/v1/chat/completions` - OpenAI Chat Completions compatibility wrapper
|
|
||||||
- `/v1/models` - List available models
|
|
||||||
|
|
||||||
- **Key Features for Development**:
|
|
||||||
- **Responses API Format**: Structured output with reasoning, function_call, and message items
|
|
||||||
- **Parameter Validation**: Temperature, reasoning effort levels, max tokens, stop sequences
|
|
||||||
- **Conversation History**: Hybrid client/server approach with auto-generated IDs
|
|
||||||
- **Context Management**: Token counting and window trimming
|
|
||||||
- **Streaming**: Real-time SSE streaming with stop sequence and max token enforcement
|
|
||||||
- **Error Handling**: Custom exception types (RateLimitError, ContextLengthError)
|
|
||||||
- **Tool Calling**: PydanticAI tool integration with permanent tools
|
|
||||||
- **Testing**: Comprehensive test suite with mocks and real Ollama integration
|
|
||||||
|
|
||||||
## FastAPI Best Practices
|
|
||||||
|
|
||||||
This project follows best practices from [github.com/zhanymkanov/fastapi-best-practices](https://github.com/zhanymkanov/fastapi-best-practices)
|
|
||||||
|
|
||||||
### Project Structure
|
|
||||||
|
|
||||||
**Domain-Based Organization**: Code is organized by domain/feature rather than by file type:
|
|
||||||
|
|
||||||
```
|
|
||||||
src/
|
|
||||||
├── agents/ # Agent interface and implementations
|
|
||||||
│ ├── base.py # Abstract AgentInterface
|
|
||||||
│ ├── lorem_tester.py # Full-featured mock agent
|
|
||||||
│ ├── tatlock.py # Placeholder for real agent
|
|
||||||
│ └── registry.py # ModelRegistry for agent management
|
|
||||||
├── responses/ # Responses API domain (PRIMARY)
|
|
||||||
│ ├── router.py # POST /v1/responses endpoint
|
|
||||||
│ ├── schemas.py # Request/response models with validators
|
|
||||||
│ ├── service.py # Response generation logic
|
|
||||||
│ ├── streaming.py # SSE streaming coordinator
|
|
||||||
│ ├── history.py # Conversation history management
|
|
||||||
│ └── context.py # Context window and token management
|
|
||||||
├── chat/ # Chat Completions domain (WRAPPER)
|
|
||||||
│ ├── router.py # POST /v1/chat/completions endpoint
|
|
||||||
│ ├── schemas.py # Chat request/response models
|
|
||||||
│ ├── service.py # Wraps Responses API, converts to <think> tags
|
|
||||||
│ ├── constants.py # Chat constants (roles, finish reasons)
|
|
||||||
│ └── __init__.py
|
|
||||||
├── models/ # Models listing domain
|
|
||||||
│ ├── router.py # GET /v1/models endpoint
|
|
||||||
│ ├── schemas.py # Model schemas
|
|
||||||
│ ├── service.py # Accesses ModelRegistry
|
|
||||||
│ └── __init__.py
|
|
||||||
├── core/ # Shared utilities
|
|
||||||
│ ├── config.py # Global configuration (BaseSettings)
|
|
||||||
│ ├── models.py # Custom base Pydantic models
|
|
||||||
│ ├── exceptions.py # Custom exceptions (RateLimitError, etc.)
|
|
||||||
│ ├── dependencies.py # Shared dependencies
|
|
||||||
│ └── router.py # Core routes (health, root)
|
|
||||||
├── ollama/ # Ollama client layer (not yet integrated)
|
|
||||||
│ ├── client.py # Async Ollama HTTP client
|
|
||||||
│ └── schemas.py # Ollama API models
|
|
||||||
└── main.py # Application factory & configuration
|
|
||||||
```
|
|
||||||
|
|
||||||
**Key Architectural Principles**:
|
|
||||||
- **Single Source of Truth**: Responses API handles all generation logic
|
|
||||||
- **Wrapper Pattern**: Chat Completions wraps Responses API without duplicating code
|
|
||||||
- **Agent Abstraction**: AgentInterface defines contract for all models
|
|
||||||
- **Domain Separation**: Each domain has its own router, schemas, service
|
|
||||||
- **Service Layer**: Business logic in services, not routers
|
|
||||||
- **Type Safety**: Pydantic models for ALL request/response validation
|
|
||||||
- **Async First**: All I/O operations use async/await
|
|
||||||
|
|
||||||
### Async/Await Best Practices
|
|
||||||
|
|
||||||
**Critical Understanding**: FastAPI handles sync and async routes differently:
|
|
||||||
|
|
||||||
- **Async routes** (`async def`): Called directly in event loop
|
|
||||||
- Use ONLY for non-blocking operations
|
|
||||||
- Perfect for `await httpx.get()`, database queries, file I/O
|
|
||||||
- **NEVER** use blocking calls like `time.sleep()` - this blocks entire server
|
|
||||||
|
|
||||||
- **Sync routes** (`def`): Run in thread pool
|
|
||||||
- Use for CPU-intensive work or blocking SDKs
|
|
||||||
- Blocking I/O won't freeze the event loop
|
|
||||||
- Example: `time.sleep(10)` is safe here
|
|
||||||
|
|
||||||
**Example**:
|
|
||||||
```python
|
|
||||||
@router.get("/terrible")
|
|
||||||
async def terrible():
|
|
||||||
time.sleep(10) # ❌ BLOCKS ENTIRE SERVER
|
|
||||||
|
|
||||||
@router.get("/good")
|
|
||||||
def good():
|
|
||||||
time.sleep(10) # ✅ Runs in thread pool
|
|
||||||
|
|
||||||
@router.get("/perfect")
|
|
||||||
async def perfect():
|
|
||||||
await asyncio.sleep(10) # ✅ Non-blocking async
|
|
||||||
```
|
|
||||||
|
|
||||||
**For CPU-intensive tasks**: Use separate worker processes (not threads) due to Python's GIL.
|
|
||||||
|
|
||||||
### Pydantic Configuration
|
|
||||||
|
|
||||||
**Custom Base Model**: All schemas inherit from `CustomBaseModel` for consistent behavior:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# src/core/models.py
|
|
||||||
class CustomBaseModel(BaseModel):
|
|
||||||
model_config = ConfigDict(
|
|
||||||
json_encoders={datetime: datetime_to_iso_str},
|
|
||||||
populate_by_name=True,
|
|
||||||
use_enum_values=True,
|
|
||||||
validate_assignment=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
def serializable_dict(self, **kwargs):
|
|
||||||
"""Return dict with only JSON-serializable fields."""
|
|
||||||
return jsonable_encoder(self.model_dump(**kwargs))
|
|
||||||
```
|
|
||||||
|
|
||||||
**Benefits**:
|
|
||||||
- Consistent datetime serialization across all responses
|
|
||||||
- Alias support for field name flexibility
|
|
||||||
- Easy JSON encoding for logging/debugging
|
|
||||||
|
|
||||||
**Decoupled Settings**: Split configuration by domain instead of one monolithic file:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# src/core/config.py - Global settings
|
|
||||||
class Config(BaseSettings):
|
|
||||||
DATABASE_URL: PostgresDsn
|
|
||||||
ENVIRONMENT: Environment
|
|
||||||
|
|
||||||
# src/chat/config.py - Chat-specific settings
|
|
||||||
class ChatConfig(BaseSettings):
|
|
||||||
MAX_TOKENS: int
|
|
||||||
DEFAULT_TEMPERATURE: float
|
|
||||||
```
|
|
||||||
|
|
||||||
### Dependency Injection Patterns
|
|
||||||
|
|
||||||
**Validation with Dependencies**: Use dependencies for complex validations:
|
|
||||||
|
|
||||||
```python
|
|
||||||
async def valid_post_id(post_id: UUID4) -> dict:
|
|
||||||
"""Validate post exists in database."""
|
|
||||||
post = await service.get_by_id(post_id)
|
|
||||||
if not post:
|
|
||||||
raise PostNotFound()
|
|
||||||
return post
|
|
||||||
|
|
||||||
@router.get("/posts/{post_id}")
|
|
||||||
async def get_post(post: dict = Depends(valid_post_id)):
|
|
||||||
return post # Already validated!
|
|
||||||
```
|
|
||||||
|
|
||||||
**Chaining Dependencies**: Build reusable validation layers:
|
|
||||||
|
|
||||||
```python
|
|
||||||
async def valid_owned_post(
|
|
||||||
post: dict = Depends(valid_post_id),
|
|
||||||
token_data: dict = Depends(parse_jwt_data),
|
|
||||||
) -> dict:
|
|
||||||
if post["creator_id"] != token_data["user_id"]:
|
|
||||||
raise UserNotOwner()
|
|
||||||
return post
|
|
||||||
```
|
|
||||||
|
|
||||||
**Dependency Caching**: Dependencies are cached within request scope - FastAPI only executes each dependency once per request, even if used multiple times.
|
|
||||||
|
|
||||||
### Application Factory Pattern
|
|
||||||
|
|
||||||
Main.py uses factory pattern for testability and configuration:
|
|
||||||
|
|
||||||
```python
|
|
||||||
def create_application() -> FastAPI:
|
|
||||||
"""Create and configure FastAPI app."""
|
|
||||||
app = FastAPI(title=config.APP_NAME)
|
|
||||||
|
|
||||||
# Add middleware
|
|
||||||
app.add_middleware(CORSMiddleware, ...)
|
|
||||||
|
|
||||||
# Register exception handlers
|
|
||||||
register_exception_handlers(app)
|
|
||||||
|
|
||||||
# Include routers
|
|
||||||
app.include_router(chat_router, prefix="/v1")
|
|
||||||
|
|
||||||
return app
|
|
||||||
|
|
||||||
app = create_application()
|
|
||||||
```
|
|
||||||
|
|
||||||
## Development Guidelines
|
|
||||||
|
|
||||||
### Git Workflow
|
|
||||||
|
|
||||||
**IMPORTANT**: Do NOT handle git commits or pushes automatically. Wait for explicit user instruction before:
|
|
||||||
- Running `git add`
|
|
||||||
- Running `git commit`
|
|
||||||
- Running `git push`
|
|
||||||
- Creating or pushing tags
|
|
||||||
|
|
||||||
The user will manage git operations themselves unless they specifically request assistance.
|
|
||||||
|
|
||||||
### Server Logs and Debugging
|
|
||||||
|
|
||||||
**Development Mode Logging**: When the server is started using `./wakeup.sh`, logs are written to `logs/server.log`. This file is:
|
|
||||||
- Cleared on each server startup (fresh logs every time)
|
|
||||||
- Written in real-time as the server runs
|
|
||||||
- Already gitignored (won't be committed)
|
|
||||||
|
|
||||||
**Accessing Logs**: You can read the log file at any time while the server is running:
|
|
||||||
```bash
|
|
||||||
# View current logs
|
|
||||||
cat logs/server.log
|
|
||||||
|
|
||||||
# Follow logs in real-time
|
|
||||||
tail -f logs/server.log
|
|
||||||
|
|
||||||
# Search logs
|
|
||||||
grep "ERROR" logs/server.log
|
|
||||||
```
|
|
||||||
|
|
||||||
This is useful for debugging issues, monitoring API calls, and understanding server behavior during development.
|
|
||||||
|
|
||||||
### Code Structure Guidelines
|
|
||||||
- Use async/await for ALL I/O operations (database, HTTP, file access)
|
|
||||||
- Use sync (def) for blocking SDKs or CPU-intensive work
|
|
||||||
- Implement proper error handling and logging
|
|
||||||
- Follow dependency injection for validation and shared resources
|
|
||||||
- Use Pydantic models for ALL request/response validation
|
|
||||||
- Keep business logic in service modules, not routers
|
|
||||||
- Domain-based project structure (not file-type based)
|
|
||||||
|
|
||||||
### Security Considerations
|
|
||||||
- Validate all inputs using Pydantic models
|
|
||||||
- Use environment variables for sensitive configuration
|
|
||||||
- Keep dependencies updated and CVE-checked
|
|
||||||
- Minor version locking for supply chain protection
|
|
||||||
- Consider rate limiting for production deployment
|
|
||||||
- Plan for authentication/API keys when needed
|
|
||||||
|
|
||||||
### Testing Approach
|
|
||||||
- Write integration tests for API endpoints
|
|
||||||
- Test streaming functionality with appropriate timeouts
|
|
||||||
- Use pytest-asyncio for async test support
|
|
||||||
- Validate OpenAI API compatibility in tests
|
|
||||||
- Test both mock and real LLM integrations
|
|
||||||
- Cover main application (CORS, exception handlers, lifespan)
|
|
||||||
- Test wrapper layers (chat completions, etc.)
|
|
||||||
- Include tool functionality tests
|
|
||||||
|
|
||||||
### Configuration Management
|
|
||||||
- Use `.env` files for local development
|
|
||||||
- Document all environment variables in README
|
|
||||||
- Provide sensible defaults where possible
|
|
||||||
- Use BaseSettings from pydantic-settings
|
|
||||||
- Support both local and container-based configuration
|
|
||||||
|
|
||||||
## Common Patterns
|
|
||||||
|
|
||||||
### Streaming Response Pattern
|
|
||||||
|
|
||||||
Example from `src/chat/router.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
from sse_starlette.sse import EventSourceResponse
|
|
||||||
from fastapi import FastAPI
|
|
||||||
|
|
||||||
async def event_generator():
|
|
||||||
# Currently yields mock lorem ipsum chunks
|
|
||||||
# Future: Stream from Ollama/PydanticAI
|
|
||||||
yield {"data": chunk.model_dump_json()}
|
|
||||||
yield {"data": "[DONE]"}
|
|
||||||
|
|
||||||
@app.post("/stream")
|
|
||||||
async def stream():
|
|
||||||
return EventSourceResponse(event_generator())
|
|
||||||
```
|
|
||||||
|
|
||||||
### PydanticAI Agent Pattern
|
|
||||||
|
|
||||||
When implementing agents with PydanticAI and Ollama:
|
|
||||||
|
|
||||||
```python
|
|
||||||
from pydantic_ai import Agent
|
|
||||||
|
|
||||||
agent = Agent(
|
|
||||||
'ollama:mistral-nemo', # Target model
|
|
||||||
# Configuration here
|
|
||||||
)
|
|
||||||
|
|
||||||
# Use the agent
|
|
||||||
result = await agent.run('Your prompt')
|
|
||||||
```
|
|
||||||
|
|
||||||
### OpenAI-Compatible Response Format
|
|
||||||
|
|
||||||
Example schema from `src/chat/schemas.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
{
|
|
||||||
"id": "chatcmpl-123",
|
|
||||||
"object": "chat.completion.chunk",
|
|
||||||
"created": 1234567890,
|
|
||||||
"model": "mistral-nemo:latest",
|
|
||||||
"choices": [{
|
|
||||||
"index": 0,
|
|
||||||
"delta": {"content": "response"},
|
|
||||||
"finish_reason": None
|
|
||||||
}]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
### PydanticAI Tool Registration Pattern
|
|
||||||
|
|
||||||
Tools are registered with PydanticAI agents using decorators. See `src/agents/tatlock.py` for examples:
|
|
||||||
|
|
||||||
```python
|
|
||||||
from pydantic_ai import Agent, RunContext
|
|
||||||
|
|
||||||
# After creating the agent
|
|
||||||
@agent.tool
|
|
||||||
def tool_name(ctx: RunContext[None], param: str) -> str:
|
|
||||||
"""
|
|
||||||
Tool description that the LLM sees.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
param: Parameter description
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Result description
|
|
||||||
"""
|
|
||||||
return result
|
|
||||||
```
|
|
||||||
|
|
||||||
**Tool Implementation Guidelines**:
|
|
||||||
- Keep tools in `src/agents/tools.py` for reusability
|
|
||||||
- Use clear, descriptive docstrings (LLM reads these)
|
|
||||||
- Include parameter descriptions in docstrings
|
|
||||||
- Handle errors gracefully and return error messages as strings
|
|
||||||
- For async operations, declare the tool function as `async def`
|
|
||||||
- Test tools independently before integration
|
|
||||||
|
|
||||||
**Example Tool Module** (`src/agents/tools.py`):
|
|
||||||
```python
|
|
||||||
def calculate(expression: str) -> str:
|
|
||||||
"""Safe calculator implementation."""
|
|
||||||
try:
|
|
||||||
# Implementation
|
|
||||||
return str(result)
|
|
||||||
except Exception as e:
|
|
||||||
return f"Error: {str(e)}"
|
|
||||||
|
|
||||||
async def search_web(query: str) -> str:
|
|
||||||
"""Web search via SearXNG."""
|
|
||||||
async with httpx.AsyncClient() as client:
|
|
||||||
# Implementation
|
|
||||||
return formatted_results
|
|
||||||
```
|
|
||||||
|
|
||||||
## Update Policy
|
|
||||||
|
|
||||||
This document should be updated when:
|
|
||||||
- New development patterns are established
|
|
||||||
- Package versions are upgraded
|
|
||||||
- Major architectural changes occur
|
|
||||||
- New best practices are identified
|
|
||||||
|
|
||||||
Last updated: 2025-12-06 (Tools integration)
|
|
||||||
+755
-1
@@ -7,6 +7,730 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- `make setup` now ends with a `pytest --collect-only` pass so a broken environment
|
||||||
|
(missing or mismatched dependency) fails the target itself instead of exiting 0
|
||||||
|
and surfacing later as a confusing test failure (T-47)
|
||||||
|
|
||||||
|
## [2.4.3] - 2026-08-08
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Steward routing no longer triggers on words inside its own explanation. Capability
|
||||||
|
extraction reads the declared `DELEGATE:` line instead of substring-matching
|
||||||
|
capability domains across the whole response, where ordinary English routed
|
||||||
|
requests — "description" contains the housekeeper domain "script", "acknowledge"
|
||||||
|
contains "knowledge" and "know". A spurious capability meant a real agent call,
|
||||||
|
including web searches, on queries that needed none.
|
||||||
|
|
||||||
|
## [2.4.2] - 2026-07-19
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- Container crash-loop on fresh builds: cap `opentelemetry-api` below 1.44,
|
||||||
|
which removed the private `_events` module that pydantic-ai 1.27 imports
|
||||||
|
|
||||||
|
## [2.4.1] - 2026-07-19
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Container-name network defaults** - `SEARXNG_HOST`, `LIBRARY_DESK_HOST`, and `CORE_API_HOST` now default to docker container names on the docker-dataplane network (`http://searxng:8080`, `http://library-desk:8089`, `http://core-api:8083`) instead of host `localhost` ports, ahead of the loopback port rebinding; this also fixes `CORE_API_HOST` pointing at port 8090 (the Scheduler's host port) rather than Core-API's 8083. `scripts/test_housekeeper.sh` now reaches Core-API via `localhost:8083` instead of the LAN IP. Local development against host-published ports still works via `.env` overrides
|
||||||
|
|
||||||
|
## [2.4.0] - 2026-07-14
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- **Dead delegation stack** - deleted the duplicate, never-wired coordination layer so exactly ONE delegation implementation remains (`src/agents/delegation.py`): `src/agents/coordination.py` (`CoordinationEngine`, its own `delegate_to_librarian`, `AGENT_EXECUTORS`/`AGENT_STREAM_EXECUTORS`), the broken-by-design `run_librarian_stream` path it used (Ollama streaming + tool call bug), the `stream_delegate_to_*` wrappers with their never-parsed `__DELEGATION_RESULT__` marker, and `HouseholdRegistry.get_streaming_delegation_tools()` (no callers)
|
||||||
|
- **Orphaned agent protocol models** - `src/agents/protocol.py` now contains only the live `AgentError`; the coordination wire protocol it carried (`AgentRequest`, `AgentResponse`, `DelegationIntent`, `CoordinationResult`, `DelegationReason`, `TaskComplexity`, `ToolCallRecord`, `AgentTimeoutError`, `AgentUnavailableError`, `DelegationError`) had no importer left outside its own tests after the coordination stack removal
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Test-suite tenant guard** - `tests/conftest.py` hard-fails the whole pytest session (exit code 1, zero tests run) if the effective tenant resolves to the production tenant `jpmschweitzer`, mirroring the guard library-desk applies on its side. Suite-level assertions pin that the session runs under `llm_tester` namespaces (Qdrant `memories_llm_tester`, Redis `session:llm_tester:*`), and the e2e isolation constants now derive from the shared `TEST_TENANT`/`PRODUCTION_TENANT` config constants instead of string literals
|
||||||
|
- **Explicit tenant on every library-desk request** - the librarian client now resolves and sends the `user` parameter explicitly on every request (library-desk is removing its server-side default; a missing user would 422). The content extraction endpoints now carry the tenant too, `search_web` no longer falls back to a phantom `tatlock-librarian` user, and a client-level assertion rejects an empty/whitespace tenant before any bytes hit the wire. A parametrized sweep pins the wire contract for all 15 tenant-scoped client methods
|
||||||
|
- **Tenant isolation guard** - non-production environments (development/testing) now FORCE the effective tenant to the reserved test tenant `llm_tester` (only `llm_tester` itself or a `test_`-prefixed override is accepted), regardless of `DEFAULT_USER` misconfiguration, at both config resolution and request-context resolution (`get_user()`). Startup refuses (clear error) when a non-production environment is explicitly configured with the production tenant `jpmschweitzer`, and one loud startup log line states the effective/forced tenant
|
||||||
|
|
||||||
|
- **Conversation context for experts + real-time think messages** - direct delegation (streaming and non-streaming) now passes a trimmed conversation history (last 6 turns) as expert context, so follow-up questions keep their referent; `_stream_direct_delegation` is now an async generator, so butler think messages ("Allow me to consult the archives, sir.") stream BEFORE the research runs instead of after it completes
|
||||||
|
- **Bounded retries and connection reuse for library-desk** - GETs and the read-only `POST /query/*` and `POST /rag/search` endpoints retry once (2 attempts, short backoff) on transport errors and retryable 5xx; wiki writes are never retried. The client now honors `LIBRARY_DESK_TIMEOUT` instead of hardcoded 60s/30s, a librarian run holds one shared HTTP connection instead of constructing a client per tool call, and read tools raise `ModelRetry` on transient HTTP errors so the agent's retry budget engages
|
||||||
|
- **One librarian timeout budget** - new `LIBRARIAN_TIMEOUT` (default 180s) enforced with `asyncio.wait_for` inside `delegate_to_librarian`, capping the previously uncapped live paths (steward direct delegation and streaming). The Ollama provider's AsyncOpenAI client now carries an explicit `OLLAMA_TIMEOUT` instead of the SDK's ~600s default, and the contradictory unused 60s default in `AgentRequest.timeout_seconds` was removed (None defers to the configured budget)
|
||||||
|
- **Search degradation signaling** - The librarian client parses `source_counts` (plus the additive `source_status`/`degraded` fields when a newer library-desk sends them; absence is tolerated), and `hybrid_search` appends a one-line coverage note when a search is degraded or an enabled source leg contributed nothing, so outages are visible to the model and the user. When `source_status` is present it is used exclusively; without it, count-absence is only inferred for the optional legs the request explicitly enabled (web/documents/volatile) - never the always-on vector/graph legs, whose absence from the top-N counts is normal ranking behavior, so healthy searches no longer emit warnings
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Clearing all wiki-page tags is possible again** - the Ollama-safe empty-list sentinel in `update_wiki_page` means "leave unchanged", which made it impossible to remove all tags; passing exactly `["__CLEAR__"]` now sends an empty tag list to library-desk (documented in the tool docstring for the local model)
|
||||||
|
- **Text-delegation fallback pairs results strictly** - the parallel branch now verifies `asyncio.gather` returned one result per parsed delegation (`zip(..., strict=True)`); a count mismatch fails loudly with a curated apology instead of silently attributing outputs to the wrong agent
|
||||||
|
- **Ollama-safe librarian tool schemas** - `update_wiki_page` and `smart_create_wiki_page` no longer use `X | None` parameters (Ollama's OpenAI-compatible API mishandles `anyOf[X, null]`); empty-string/empty-list sentinels are translated to `None` inside the tools, matching the biographer pattern. A snapshot test pins every librarian tool schema to contain no nullable `anyOf`
|
||||||
|
- **Honest expert failures** - `run_librarian` now raises a structured `AgentError` instead of returning error text as if it were research output, so delegation correctly reports `success=False` and the streaming error branch is reachable. Failures surface to the user as curated butler-toned sentences; exception detail (including internal URLs) stays in the logs only. Librarian tool errors no longer leak `str(e)` into synthesis
|
||||||
|
|
||||||
|
- **HybridRAG response mapping** - The librarian client now parses the field names library-desk actually returns (`source_type`/`sources`, `rrf_score`, `context`, per-item `related_dossiers`, synonyms nested in the `keywords` dict); previously every result rendered as "unknown (score: 0.00)". Source icons now key off the per-item `sources` list. Requests no longer send zero limits (the service rejects them with 422); legs are disabled via `enable_*` flags. Pinned by a contract test against a recorded live response (`tests/agents/librarian/fixtures/`)
|
||||||
|
|
||||||
|
## [2.3.0] - 2026-07-13
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Local-first backend (claudification rollback)** - Ollama/gemma4 is now the primary backend; Claude remains as fallback. `PREFER_CLOUD_BACKEND` defaults to `false`, Claude is used automatically when the Ollama startup health check fails, and the Steward retries mid-request failures on the other backend in both directions
|
||||||
|
- **Default Claude model `claude-sonnet-5`** - `claude-sonnet-4-20250514` was retired by Anthropic on 2026-06-15 and would 404, leaving the fallback dead
|
||||||
|
- **Dedicated orchestration prompt** - `orchestrate_tool_calls()` now uses a terse tool-execution prompt (`TATLOCK_ORCHESTRATION_PROMPT`); the butler persona prompt suppressed gemma4 tool calling (the model reasoned about the calculator, then answered from memory with wrong arithmetic). Synthesis keeps the persona prompt, so user-visible voice is unchanged
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Startup crash with broken anthropic package** - Anthropic SDK imports in the model selector are now lazy, so an incompatible `anthropic` install degrades to Ollama-only operation instead of crashing the app at import time (root cause of the production outage since April)
|
||||||
|
- **Claude Sonnet 5 rejects sampling parameters** - removed `temperature` from the Steward's direct Claude call and made the Housekeeper's temperature setting backend-conditional via `get_sampling_settings()`
|
||||||
|
- **Pin `anthropic>=0.77,<1.0`** - the April image resolved an anthropic version incompatible with pydantic-ai 1.27
|
||||||
|
- **Steward timeout configurable** - new `STEWARD_TIMEOUT` (default 60s) replaces the hardcoded 30s, which gemma4 chronically exceeded (~35s warm analysis), causing every request to fail or fall back
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Ollama startup health check** - verifies the server is reachable and `OLLAMA_DEFAULT_MODEL` is pulled; feeds backend resolution and `get_model_info()`
|
||||||
|
- **Contract tests** (`tests/contracts/`, `make test-contracts`) - wire-level tests that send the raw requests the code sends to Ollama (native + OpenAI-compat tool calling), Anthropic (including the pinned temperature-rejection contract), Qdrant, SearXNG, library-desk, and Redis; unreachable services skip, wrong response shapes fail
|
||||||
|
- **Backend resolution unit tests** (`tests/anthropic/`)
|
||||||
|
|
||||||
|
## [2.2.0] - 2026-04-04
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Switch default Ollama model to gemma4:e2b** - Replaces mistral-nemo as the local LLM backend; gemma4:e2b has native function calling support, faster tool calling (2-4s vs 15-20s), better parameter accuracy on word problems, and uses less VRAM (8GB vs 9.2GB)
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Tool calling benchmark script** (`scripts/benchmark_tool_calling.py`) - Compares tool calling accuracy and latency across Ollama models via the Tatlock API
|
||||||
|
|
||||||
|
## [2.1.0] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Streaming SSE compatibility with Open WebUI** - Switch from `exclude_none=True` to `exclude_unset=True` for SSE chunk serialization; `exclude_none` was too aggressive — it stripped `finish_reason: null` from intermediate chunks (which OpenAI includes), while `exclude_unset` correctly omits only fields never passed to the constructor (like `reasoning_content` on content-only chunks) while preserving explicitly-set `finish_reason: null`
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Project structure consolidation** - Moved documentation to `docs/`, consolidated all config into `pyproject.toml`, replaced `wakeup.sh`/`pytest.ini`/`requirements*.txt` with `Makefile` + `pyproject.toml`
|
||||||
|
- **CI test gate** - Unit tests now gate release and build jobs in Gitea Actions workflow
|
||||||
|
- **Build output organization** - Tool caches in `.cache/`, generated output (coverage, logs) in `build/`
|
||||||
|
|
||||||
|
## [2.0.5] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Streaming JSON compatibility** - Exclude null fields from streaming chunks using `exclude_none=True`; OpenAI's API omits null fields entirely, and including them (e.g., `content: null`, `reasoning_content: null`) caused parsing issues in Open WebUI
|
||||||
|
|
||||||
|
## [2.0.4] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Open WebUI streaming compatibility** - Replaced `sse_starlette` `EventSourceResponse` with plain `StreamingResponse` for chat completions; `sse_starlette` added `\r\n` line endings and extra SSE fields that Open WebUI couldn't parse
|
||||||
|
|
||||||
|
## [2.0.3] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Steward analysis leaking into responses** - Removed internal routing analysis (`DELEGATE: tatlock_core...`) from user-visible reasoning in both streaming and non-streaming paths
|
||||||
|
|
||||||
|
## [2.0.2] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **tool_choice format incompatibility** - Removed `extra_body` tool_choice hack for Claude backend; PydanticAI handles tool_choice natively for Anthropic, preventing infinite tool call loops
|
||||||
|
- **CI trigger** - Changed workflow trigger from `release:published` to `push:tags:v[0-9]*`
|
||||||
|
|
||||||
|
## [2.0.1] - 2026-02-05
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Expert agent registration failure** - `AnthropicModel` does not accept `api_key` directly; now passes it via `AnthropicProvider`
|
||||||
|
|
||||||
|
## [2.0.0] - 2026-02-05
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Claude backend support (Claudification Phase 1)** - All agents now prefer Claude over Ollama
|
||||||
|
- New `src/anthropic/` module with model selector and health check
|
||||||
|
- `get_model()` factory returns Claude if available, Ollama as fallback
|
||||||
|
- Startup health check caches Claude API availability
|
||||||
|
- Configuration: `ANTHROPIC_API_KEY`, `ANTHROPIC_MODEL`, `PREFER_CLOUD_BACKEND`
|
||||||
|
- 200k token context when using Claude backend
|
||||||
|
|
||||||
|
- **Steward dual-backend support** - Direct API calls to Claude or Ollama
|
||||||
|
- `_call_claude()`: Anthropic Messages API path
|
||||||
|
- `_call_ollama()`: Existing Ollama generate API path (preserved)
|
||||||
|
- Automatic fallback: if Claude call fails mid-request, retries with Ollama
|
||||||
|
|
||||||
|
- **Claudification project tracking** - `PROJECT_CLAUDIFICATION.md` with Phase 1/2 roadmap
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **All PydanticAI agents refactored to use `get_model()`**:
|
||||||
|
- Tatlock (6 instantiation locations)
|
||||||
|
- Librarian
|
||||||
|
- Biographer
|
||||||
|
- Housekeeper
|
||||||
|
- **`initialize_application()` is now async** - Supports async Claude health check at startup
|
||||||
|
- **Dependencies**: `pydantic-ai-slim[openai,anthropic]` replaces `pydantic-ai-slim[openai]`
|
||||||
|
- **Startup logging** now includes backend selection info (claude/ollama)
|
||||||
|
- **Agent creation logging** now includes backend and model info
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- Stale `tests/core/test_benchmarks.py` (benchmark system was removed in v1.10.0)
|
||||||
|
|
||||||
|
## [1.11.0] - 2025-12-30
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Paperless document integration** - HybridRAG now includes indexed PDFs and scanned documents from Paperless-ngx
|
||||||
|
- New `include_documents` parameter in `hybrid_search` tool
|
||||||
|
- 📑 icon for document sources in search results
|
||||||
|
- Librarian prompt updated with document awareness
|
||||||
|
|
||||||
|
- **Volatile cache integration** - HybridRAG now includes pre-fetched real-time data
|
||||||
|
- New `include_volatile` parameter in `hybrid_search` tool
|
||||||
|
- ⚡ icon for volatile sources in search results
|
||||||
|
- Supports weather, forecast, news, stock, crypto, sun, air_quality namespaces
|
||||||
|
- Librarian prompt updated with volatile cache awareness (user-configured items only)
|
||||||
|
|
||||||
|
- **Biographer routing in Steward** - Personal memory queries now correctly route to The Biographer
|
||||||
|
- Added explicit routing rules for "where do I live", "what car do I drive", etc.
|
||||||
|
- Added biographer delegation examples to Steward prompt
|
||||||
|
- Location keywords ("live", "where", "home") now trigger profile pre-fetch
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **LibraryDeskClient.hybrid_search** - Now passes full config including `document_limit`, `volatile_limit`, and enable flags
|
||||||
|
- **Steward guidelines** - Clarified that research queries about TOPICS go to Librarian, queries about USER go to Biographer
|
||||||
|
|
||||||
|
## [1.10.1] - 2025-12-23
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Tatlock's excessive apologizing** - Strengthened personality prompt to prevent unnecessary apologies after successful Librarian delegations. Added explicit "do NOT apologize" instructions to both system prompt and synthesis prompt.
|
||||||
|
|
||||||
|
## [1.10.0] - 2025-12-22
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Lightweight Request Tracing
|
||||||
|
- **JSON-based tracing system** for local development debugging
|
||||||
|
- Captures full request flow through multi-agent architecture
|
||||||
|
- `Trace` and `Span` dataclasses with automatic timing and nesting
|
||||||
|
- ContextVar-based propagation for async-safe tracing
|
||||||
|
- `trace_span` async context manager for clean instrumentation
|
||||||
|
- Traces written to `logs/traces/{trace_id}.json`
|
||||||
|
- Enabled via `DEBUG=true` environment variable
|
||||||
|
- **Trace Viewer UI** (`logs/traces/viewer.html`)
|
||||||
|
- Standalone HTML viewer with timeline visualization
|
||||||
|
- Filter by status, search by request text
|
||||||
|
- Expandable span details with prompts and responses
|
||||||
|
- **Tracing REST API** (`/traces`)
|
||||||
|
- `GET /traces` - Serve trace viewer UI
|
||||||
|
- `GET /traces/list` - List available traces with filtering
|
||||||
|
- `GET /traces/{trace_id}` - Retrieve specific trace JSON
|
||||||
|
- Only available when `DEBUG=true`
|
||||||
|
- **Full pipeline instrumentation**
|
||||||
|
- Router-level trace start/end with context management
|
||||||
|
- Steward analysis spans in preprocessing
|
||||||
|
- Tatlock orchestrate/synthesize spans
|
||||||
|
- Expert delegation spans (librarian/biographer/housekeeper)
|
||||||
|
- Tool-level spans extracted from PydanticAI messages
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Replaced Redis benchmarks with file-based tracing** - Simpler, more useful for debugging
|
||||||
|
- **Context management moved to service layer** - Router simplified, context set in response service
|
||||||
|
- **Server binds to all interfaces** - `wakeup.sh` now uses `0.0.0.0` for network access
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- **Redis benchmark system** (`src/core/benchmarks.py`)
|
||||||
|
- `ENABLE_BENCHMARKS` config setting
|
||||||
|
- `REDIS_BENCHMARK_DB` config setting
|
||||||
|
- `redis_url` property (kept `redis_memory_url`)
|
||||||
|
- Benchmark recording in Steward service and tool tracking
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Librarian fabrication prevention** - Added explicit instructions to never invent data when tools fail or sources are unavailable
|
||||||
|
|
||||||
|
## [1.9.0] - 2025-12-18
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Housekeeper prompt optimization** - Rewrote system prompt for Mistral-Nemo function calling with negative constraints, step-by-step process, and explicit entity ID format guidance
|
||||||
|
- **Housekeeper temperature setting** - Set temperature to 0.1 for deterministic tool calling behavior
|
||||||
|
- **Device list room group priority** - Room groups now appear first in `list_devices` output with `[ROOM GROUP]` marker to address positional bias
|
||||||
|
- **Tool docstring improvements** - Updated turn_on/turn_off/toggle with explicit `entity_id=` parameter examples
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Housekeeper optimization findings** - Added `docs/housekeeper-optimization-findings.md` documenting the experiment journey from 0% to 100% success rate
|
||||||
|
- **Housekeeper test script** - Added `scripts/test_housekeeper.sh` for room group detection regression testing
|
||||||
|
|
||||||
|
## [1.8.6] - 2025-12-17
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Housekeeper API paths** - Updated all client endpoints to use `/housekeeping/` prefix to match core-api routes
|
||||||
|
- **Housekeeper entity hallucination** - Improved system prompt with critical rule requiring `list_devices()` before any control action to prevent guessing entity IDs
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **Housekeeping API spec** - Added `docs/housekeeping-api-spec.md` documenting the core-api home automation interface
|
||||||
|
|
||||||
|
## [1.8.5] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Redis benchmark boolean storage** - Convert booleans to strings for Redis `hset` (Redis doesn't accept bool type directly)
|
||||||
|
- **Tool tracking capability matching** - `delegate_to_librarian` now correctly recognized as using "librarian" capability when checking Steward recommendations
|
||||||
|
- **E2E test fixture scope** - Fixed pytest-asyncio ScopeMismatch error by using `loop_scope="module"` for module-scoped async fixtures
|
||||||
|
|
||||||
|
## [1.8.4] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Remove `<think>` wrappers from think messages** - Messages in `reasoning_content` should be plain text
|
||||||
|
- Removed `<think>` wrappers from delegation.py household think messages
|
||||||
|
- Removed `<think>` wrappers from orchestration.py status messages
|
||||||
|
- Think messages now appear cleanly in Open WebUI's reasoning block
|
||||||
|
|
||||||
|
## [1.8.3] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Open WebUI streaming rendering** - Use `reasoning_content` field for thinking (DeepSeek R1 format) instead of `<think>` tags in `content`
|
||||||
|
- Open WebUI now renders thinking as proper collapsible blocks instead of broken HTML
|
||||||
|
|
||||||
|
## [1.8.2] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **HybridRAG keywords schema mismatch** - library-desk now returns `keywords` as dict with `core_keywords`, client now handles both formats
|
||||||
|
|
||||||
|
## [1.8.1] - 2025-12-16
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
#### Ollama Message Sanitization
|
||||||
|
- **Fixed `invalid message content type: <nil>` error** from Ollama
|
||||||
|
- Created custom `TatlockOllamaProvider` that sanitizes messages before sending to Ollama
|
||||||
|
- Ollama rejects assistant messages with `content: null` (tool-only messages from PydanticAI)
|
||||||
|
- Provider converts `null` content to empty string `""` for compatibility
|
||||||
|
- Updated all agents (Librarian, Biographer, Housekeeper, Tatlock) to use sanitized provider
|
||||||
|
- Added `src/ollama/provider.py` with reusable provider pattern
|
||||||
|
|
||||||
|
#### Streaming Think Message Accumulation
|
||||||
|
- **Fixed repeating think messages in frontend** (e.g., 10x "The Librarian has compiled...")
|
||||||
|
- Frontend was accumulating `ReasoningSummaryDelta` events expecting concatenation
|
||||||
|
- Added `ReasoningSummaryDone()` signal after each think message to indicate completion
|
||||||
|
- Each think slug is now treated as a complete message, not a continuation
|
||||||
|
|
||||||
|
## [1.8.0] - 2025-12-15
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
#### Steward Routing for Web Search
|
||||||
|
- Updated Steward guidelines to route web searches, weather, news → Librarian with `search_web`
|
||||||
|
- Added URL/article reading → Librarian with `read_url` to routing guidelines
|
||||||
|
- Added examples showing `search_web` and `read_url` tool usage
|
||||||
|
|
||||||
|
#### Librarian Agent Tool Registration
|
||||||
|
- Registered `search_web`, `read_url`, `read_urls_batch` tools with the Librarian PydanticAI agent
|
||||||
|
- Updated Librarian system prompt with Web Search & Content Extraction section
|
||||||
|
- Fixed tool count in agent logger (11 → 14 tools)
|
||||||
|
|
||||||
|
#### Query Enrichment Integration
|
||||||
|
- Fixed enriched query (with location/timezone context) not being passed to delegations
|
||||||
|
- Response service now uses `enriched_query` from Steward recommendation for all delegations
|
||||||
|
- Weather queries now automatically include user's stored location
|
||||||
|
|
||||||
|
#### Action Type Detection
|
||||||
|
- Added "read", "fetch", "url", "http" keywords to RESEARCH action type for Librarian
|
||||||
|
- Ensures proper think messages for URL reading tasks
|
||||||
|
|
||||||
|
## [1.7.0] - 2025-12-15
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Web Search Migration to Librarian
|
||||||
|
- **`search_web()`** tool in Librarian for web search via library-desk `/rag/search` endpoint
|
||||||
|
- **`read_url()`** tool for single URL content extraction via Trafilatura
|
||||||
|
- **`read_urls_batch()`** tool for parallel batch URL extraction (max 20 URLs)
|
||||||
|
- `WebSearchResult`, `WebSearchResponse` models in LibraryDeskClient
|
||||||
|
- `ContentExtractionResult`, `BatchExtractionResponse` models for content extraction
|
||||||
|
- `search_web()`, `extract_content()`, `extract_content_batch()` methods in LibraryDeskClient
|
||||||
|
- Comprehensive unit tests for new Librarian tools (`tests/agents/librarian/test_tools.py`)
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- Librarian capability updated with web search domains: "web", "url", "internet"
|
||||||
|
- Tatlock system prompt now delegates web search to Librarian
|
||||||
|
- `tatlock_core` capability reduced to computation/datetime only (no longer requires network)
|
||||||
|
|
||||||
|
### Removed
|
||||||
|
|
||||||
|
- `search_web` function from `src/agents/tatlock_core/tools.py`
|
||||||
|
- `web_search_tool` from `tatlock_core_tools` list
|
||||||
|
- `search_web` from legacy `src/agents/tools.py`
|
||||||
|
- Search tests from `tests/agents/test_tools.py` (moved to Librarian tests)
|
||||||
|
|
||||||
|
## [1.6.0] - 2025-12-15
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Two-Phase Tatlock Execution
|
||||||
|
- **Phase 1: Orchestration** - Executes tool calls and expert delegations, returns structured results
|
||||||
|
- **Phase 2: Synthesis** - Synthesizes butler-toned response from gathered results
|
||||||
|
- `orchestrate_tool_calls()` method in TatlockAgent for coordination phase
|
||||||
|
- `synthesize_from_results()` method in TatlockAgent for synthesis phase
|
||||||
|
- Guarantees butler personality in all responses by separating coordination from response generation
|
||||||
|
|
||||||
|
#### Automatic Think Slugs
|
||||||
|
- **Deterministic butler-perspective messages** during expert delegation (no LLM involved)
|
||||||
|
- `ActionType` enum: RETRIEVE, RESEARCH, CREATE, CONTROL, RECORD
|
||||||
|
- `HOUSEHOLD_THINK_MESSAGES` mapping with butler-perspective messages for all experts:
|
||||||
|
- Librarian: "Allow me to consult the archives, sir." / "I'm having the Librarian prepare a new entry."
|
||||||
|
- Biographer: "Let me consult the household records." / "I've asked the Biographer to take note, sir."
|
||||||
|
- Housekeeper: "I'm instructing the household staff now, sir." / "Allow me to inquire with the household staff."
|
||||||
|
- `_detect_action_type()` function for keyword-based action detection
|
||||||
|
- `get_think_message()` helper for retrieving appropriate messages
|
||||||
|
- Streaming delegation wrappers: `stream_delegate_to_librarian()`, `stream_delegate_to_biographer()`, `stream_delegate_to_housekeeper()`
|
||||||
|
- `STREAMING_DELEGATION_WRAPPERS` mapping in delegation.py
|
||||||
|
- `get_streaming_delegation_tools()` method in HouseholdRegistry
|
||||||
|
|
||||||
|
#### Steward Query Enrichment
|
||||||
|
- **Auto-fill user context** (location, timezone) when not specified in query
|
||||||
|
- `_build_enriched_query()` function in steward service
|
||||||
|
- Regex word boundary matching for accurate location detection (avoids false positives)
|
||||||
|
- `enriched_query` field added to `StewardRecommendation` schema
|
||||||
|
- Automatic enrichment for weather queries (location), time queries (timezone), temperature preferences
|
||||||
|
|
||||||
|
#### Documentation
|
||||||
|
- **ORCHESTRATION_SCENARIOS.md** completely rewritten with:
|
||||||
|
- Mermaid flow diagrams for two-phase execution
|
||||||
|
- 4 new Housekeeper scenarios (light control, device status, parallel delegation)
|
||||||
|
- Biographer memory recording scenario
|
||||||
|
- Complete think slug reference tables
|
||||||
|
- Action type detection tables
|
||||||
|
- Updated architecture mindmap
|
||||||
|
- **TESTING_IMPROVEMENTS.md** - LLM testing best practices for future implementation
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- `create_response_with_steward()` now uses two-phase execution
|
||||||
|
- `_direct_delegation()` routes through synthesis phase for consistent butler tone
|
||||||
|
- `_execute_single_delegation()` now supports housekeeper
|
||||||
|
- Streaming response handler integrated with think slug system
|
||||||
|
- All 326 unit tests passing
|
||||||
|
|
||||||
|
## [1.5.0] - 2025-12-15
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### The Housekeeper Agent
|
||||||
|
- **New home automation expert agent** following the Librarian pattern
|
||||||
|
- `CoreAPIClient` for communicating with core-api service (Home Assistant wrapper)
|
||||||
|
- 13 tools for home automation:
|
||||||
|
- Discovery: `list_areas`, `list_devices`, `get_device_state`
|
||||||
|
- Control: `turn_on`, `turn_off`, `toggle`
|
||||||
|
- Scenes: `list_scenes`, `activate_scene`
|
||||||
|
- Scripts: `list_scripts`, `run_script`
|
||||||
|
- Automations: `list_automations`, `toggle_automation`
|
||||||
|
- History: `get_history`
|
||||||
|
- PydanticAI agent with system prompt for home automation tasks
|
||||||
|
- `HouseholdCapability` registration with domains: lights, switches, automation, home, smart home, scene, script, device, climate, fan, cover, blinds
|
||||||
|
- `delegate_to_housekeeper()` delegation wrapper
|
||||||
|
- Config settings: `CORE_API_HOST`, `CORE_API_KEY`, `CORE_API_TIMEOUT`
|
||||||
|
|
||||||
|
#### Development Port Change
|
||||||
|
- **Dev server port changed from 8123 to 8777** to avoid conflict with Home Assistant default port
|
||||||
|
- Updated `wakeup.sh`, E2E tests, and documentation
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- All unit tests pass (421 passed, 5 xfailed)
|
||||||
|
- Housekeeper registered on startup alongside Librarian and Biographer
|
||||||
|
|
||||||
|
## [1.4.0] - 2025-12-14
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Environment-Aware Configuration
|
||||||
|
- **Auto-selected logging level**: DEBUG for development, WARNING for production
|
||||||
|
- **Auto-selected default user**: `llm_tester` for development (isolated test scope), `jpmschweitzer` for production
|
||||||
|
- Properties `effective_log_level` and `effective_default_user` in config
|
||||||
|
- User context logging at request entry with INFO level
|
||||||
|
|
||||||
|
#### Direct Delegation Bypass
|
||||||
|
- **Pure memory/librarian requests bypass Tatlock**: When Steward recommends only biographer/librarian, skip Tatlock LLM call
|
||||||
|
- `_direct_delegation()` function for immediate expert agent execution
|
||||||
|
- Reduces latency for memory-only requests
|
||||||
|
|
||||||
|
#### Text-Based Delegation Fallback
|
||||||
|
- **Parse text delegation patterns**: Handle LLM outputs like `[DELEGATE:biographer] task="..."`
|
||||||
|
- Multiple pattern support for delegation parsing
|
||||||
|
- Sequential and parallel execution with `[PARALLEL]` prefix
|
||||||
|
|
||||||
|
#### Comprehensive E2E Test Suite
|
||||||
|
- **22 new orchestration tests** in `tests/e2e/test_orchestration_e2e.py`
|
||||||
|
- `QdrantVerifier` helper class for data verification
|
||||||
|
- `assert_llm_behavior()` for flexible LLM output pattern matching
|
||||||
|
- Test classes covering:
|
||||||
|
- Memory storage and recall
|
||||||
|
- Steward delegation
|
||||||
|
- Direct delegation bypass
|
||||||
|
- User context isolation (llm_tester vs production)
|
||||||
|
- Data verification in Qdrant
|
||||||
|
- Integration health checks
|
||||||
|
- Orchestration scenarios (weather, calculator, wiki, multi-expert)
|
||||||
|
- Error handling
|
||||||
|
- Evaluation reports
|
||||||
|
- Updated `tests/e2e/README.md` with comprehensive documentation
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Unit test mocks**: Updated Steward streaming tests to mock `run_with_scoped_tools_stream` (async generator)
|
||||||
|
- **Temporal context in tests**: Tests now account for `_inject_temporal_context()` appending timestamps
|
||||||
|
- **LLM non-determinism**: Integration tests use `pytest.xfail()` for LLM-dependent assertions
|
||||||
|
- **Streaming test timeouts**: Increased timeouts (60-90s) for LLM processing time
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- All unit tests now pass (380 passed, 5 xfailed for LLM non-determinism)
|
||||||
|
- E2E tests use `llm_tester` user for isolation from production data
|
||||||
|
|
||||||
|
## [1.3.3] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Memory**: Fix Qdrant point IDs - use UUID5 instead of arbitrary strings
|
||||||
|
|
||||||
|
## [1.3.2] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Memory**: Fix biographer tool type hints for Ollama compatibility (remove `| None` union types)
|
||||||
|
|
||||||
|
## [1.3.1] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Memory**: Add biographer to delegation wrappers (was returning raw tools causing Ollama error)
|
||||||
|
- **Config**: Add Qdrant host/port to .env.example
|
||||||
|
|
||||||
|
## [1.3.0] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Memory**: Update Qdrant client to use `query_points` API (qdrant-client >= 1.10)
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Config**: Rename `REDIS_DB` to `REDIS_BENCHMARK_DB` for clarity
|
||||||
|
- **Config**: Update Redis defaults to match stack allocation (benchmark=6, memory=1)
|
||||||
|
|
||||||
|
## [1.2.5] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Dependencies**: Add missing `pydantic-settings` (not included in pydantic-ai-slim)
|
||||||
|
|
||||||
|
## [1.2.4] - 2025-12-14
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **CI**: Trigger Watchtower update after successful image push
|
||||||
|
|
||||||
|
## [1.2.3] - 2025-12-14
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **CI**: Upgrade to build-push-action@v6, disable provenance and sbom for Gitea registry
|
||||||
|
|
||||||
|
## [1.2.2] - 2025-12-13
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **CI**: Add `provenance: false` to docker/build-push-action to fix Gitea registry push
|
||||||
|
|
||||||
|
## [1.2.1] - 2025-12-13
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **Dependency slimming**: Switched from `pydantic-ai` to `pydantic-ai-slim[openai]`
|
||||||
|
- Removes unused LLM provider SDKs (anthropic, boto3, cohere, google-genai, groq, huggingface)
|
||||||
|
- Production packages: 53 (down from ~158)
|
||||||
|
- Production footprint: 178MB
|
||||||
|
- Tatlock uses Ollama via OpenAI-compatible API, so only `openai` extra is needed
|
||||||
|
- See `DEPENDENCY_SLIM.md` for rollback instructions
|
||||||
|
|
||||||
|
## [1.2.0] - 2025-12-13
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Phase F: Memory System (The Biographer)
|
||||||
|
|
||||||
|
- **Memory Infrastructure** (Phase F.1):
|
||||||
|
- `src/core/context.py`: ContextVar-based request context for async-safe user/conversation tracking
|
||||||
|
- `get_user()`, `get_conversation_id()` helpers
|
||||||
|
- `RequestContext` manager for clean setup/teardown
|
||||||
|
- `src/core/multi_tenancy.py`: User ID sanitization and collection naming
|
||||||
|
- Per-user collection pattern: `memories_{user}`
|
||||||
|
- Redis key patterns: `session:{user}:{conv}`, `entities:{user}:{conv}`
|
||||||
|
- `src/core/embeddings.py`: Ollama embedding client
|
||||||
|
- nomic-embed-text model (768 dimensions)
|
||||||
|
- `embed()`, `embed_batch()`, `health_check()` methods
|
||||||
|
- `src/core/qdrant.py`: Qdrant vector database client
|
||||||
|
- `ensure_collection()`, `upsert_memory()`, `search_memories()`, `delete_memory()`
|
||||||
|
- Type-based filtering for memory queries
|
||||||
|
- `src/core/memory_cache.py`: Redis session memory cache
|
||||||
|
- Session context with 24h TTL (db=2, separate from benchmarks)
|
||||||
|
- Recent entities tracking per conversation
|
||||||
|
|
||||||
|
- **Memory Service** (Phase F.2a):
|
||||||
|
- `src/core/memory_service.py`: Direct access layer for fast, LLM-free memory lookups
|
||||||
|
- Profile methods: `get_profile()`, `set_profile()`
|
||||||
|
- Preference methods: `get_preference()`, `set_preference()`, `get_all_preferences()`
|
||||||
|
- Fact methods: `store_fact()`, `get_fact()`
|
||||||
|
- Session context: `get_session_context()`, `set_session_context()`, `update_session_context()`
|
||||||
|
- Steward integration: `prefetch_context()` for request preprocessing
|
||||||
|
|
||||||
|
- **The Biographer Agent** (Phase F.2b):
|
||||||
|
- `src/agents/biographer/`: Household memory keeper agent
|
||||||
|
- PydanticAI agent with discreet chronicler personality
|
||||||
|
- System prompt emphasizes privacy and accurate recall
|
||||||
|
- **Biographer Tools** (`src/agents/biographer/tools.py`):
|
||||||
|
- `recall_semantic`: Semantic search for memories by meaning
|
||||||
|
- `list_memories`: Browse stored memories by type
|
||||||
|
- `store_insight`: Record new facts from conversation
|
||||||
|
- `update_profile`: Update core profile fields (name, location, timezone)
|
||||||
|
- `update_preference`: Update user preferences (units, theme)
|
||||||
|
- `forget_memory`: Remove specific memories
|
||||||
|
- **Capability Registration**:
|
||||||
|
- `BIOGRAPHER_CAPABILITY` with context domain
|
||||||
|
- Automatic registration on startup
|
||||||
|
- Low cost (vector search, minimal LLM)
|
||||||
|
|
||||||
|
- **Delegation Wrapper**:
|
||||||
|
- `delegate_to_biographer()` in `src/agents/delegation.py`
|
||||||
|
- Async delegation with error handling
|
||||||
|
|
||||||
|
- **Steward Memory Integration**:
|
||||||
|
- Memory context pre-fetch during request analysis
|
||||||
|
- Profile and preferences included in Steward's note to Butler
|
||||||
|
- Keyword-based context determination (weather → location, time → timezone)
|
||||||
|
|
||||||
|
- **Configuration**:
|
||||||
|
- `QDRANT_HOST`, `QDRANT_PORT`, `QDRANT_EMBEDDING_DIM` (768)
|
||||||
|
- `OLLAMA_EMBEDDING_MODEL` (nomic-embed-text)
|
||||||
|
- `REDIS_MEMORY_DB` (2), `REDIS_MEMORY_TTL_HOURS` (24)
|
||||||
|
|
||||||
|
- **Test Suite**:
|
||||||
|
- 34 new tests for memory system
|
||||||
|
- Biographer capability tests (15 tests)
|
||||||
|
- Memory service tests (19 tests)
|
||||||
|
|
||||||
|
- **OpenAI Standard `user` Field**:
|
||||||
|
- Added `user` field to `ResponseRequest` schema
|
||||||
|
- Request context set at API entry point
|
||||||
|
- Propagates through async calls via ContextVar
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
- Application startup now registers The Biographer with Household Registry
|
||||||
|
- Steward analysis includes memory context pre-fetch
|
||||||
|
- Librarian client methods now use `get_user()` from context (12 methods updated)
|
||||||
|
- Request router sets user/conversation context at entry
|
||||||
|
|
||||||
|
## [1.1.0] - 2025-12-11
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
#### Phase 3: Butler Orchestration (Multi-Agent Coordination)
|
||||||
|
- **The Librarian Agent**: Expert agent for research and knowledge management
|
||||||
|
- PydanticAI agent with specialized research assistant personality
|
||||||
|
- Connects to library-desk API for HybridRAG capabilities
|
||||||
|
- System prompt emphasizes fetching wiki pages before summarizing
|
||||||
|
- Streaming support via `run_librarian_stream()`
|
||||||
|
|
||||||
|
- **Library-Desk API Client** (`src/agents/librarian/client.py`):
|
||||||
|
- Async HTTP client with httpx for library-desk API integration
|
||||||
|
- HybridRAG search (vector + graph + web search)
|
||||||
|
- Wiki operations (search, get, list, create, update pages)
|
||||||
|
- Smart page creation with HybridRAG research (`POST /wiki/pages/smart-create`)
|
||||||
|
- Semantic vector search
|
||||||
|
- Knowledge graph queries (Cypher execution)
|
||||||
|
- Dossier (tag collection) browsing
|
||||||
|
- Health check endpoint
|
||||||
|
|
||||||
|
- **Librarian Tools** (`src/agents/librarian/tools.py`):
|
||||||
|
- Research tools:
|
||||||
|
- `hybrid_search`: Combined vector, graph, and web search
|
||||||
|
- `search_wiki`: Full-text wiki page search
|
||||||
|
- `get_wiki_page`: Fetch full wiki page content by ID
|
||||||
|
- `semantic_search`: Vector similarity search
|
||||||
|
- `list_dossiers`: Browse knowledge collections
|
||||||
|
- `get_dossier_pages`: Get pages in a dossier
|
||||||
|
- `explore_knowledge_graph`: Entity and relationship discovery
|
||||||
|
- `find_related_entities`: Find connected concepts
|
||||||
|
- Write tools:
|
||||||
|
- `smart_create_wiki_page`: Create page with automatic HybridRAG research (PREFERRED for topic-based creation)
|
||||||
|
- `create_wiki_page`: Create page with user-provided content
|
||||||
|
- `update_wiki_page`: Update existing page (partial updates supported)
|
||||||
|
|
||||||
|
- **Agent Communication Protocol** (`src/agents/protocol.py`):
|
||||||
|
- `AgentRequest`: Standardized task request with context and constraints
|
||||||
|
- `AgentResponse`: Response with result, reasoning, tool calls, confidence
|
||||||
|
- `DelegationIntent`: Routing intent with target agent and reason
|
||||||
|
- `CoordinationResult`: Aggregated multi-agent results
|
||||||
|
- `DelegationReason` enum: domain expertise, tool access, resource efficiency, user preference
|
||||||
|
- Error types: `AgentError`, `AgentTimeoutError`, `AgentUnavailableError`
|
||||||
|
|
||||||
|
- **Coordination Engine** (`src/agents/coordination.py`):
|
||||||
|
- `CoordinationEngine`: Multi-agent task orchestration
|
||||||
|
- Routing tasks to appropriate expert agents
|
||||||
|
- Sequential and parallel execution support
|
||||||
|
- Result aggregation from multiple agents
|
||||||
|
- Graceful error handling and degradation
|
||||||
|
- Streaming delegation support
|
||||||
|
- Convenience functions: `delegate_to_librarian()`, `delegate_to_librarian_stream()`
|
||||||
|
|
||||||
|
- **Librarian Capability Registration**:
|
||||||
|
- `LIBRARIAN_CAPABILITY` definition with research domains
|
||||||
|
- Automatic registration on application startup
|
||||||
|
- Integration with Household Registry
|
||||||
|
|
||||||
|
- **Configuration**:
|
||||||
|
- `LIBRARY_DESK_HOST`: Library-desk API URL (default: `http://localhost:8089`)
|
||||||
|
- `LIBRARY_DESK_API_KEY`: Optional API key for authentication
|
||||||
|
- `LIBRARY_DESK_TIMEOUT`: Request timeout in seconds (default: 60)
|
||||||
|
|
||||||
|
- **Test Suite**:
|
||||||
|
- 78 new tests for Phase 3 components
|
||||||
|
- Protocol model tests (requests, responses, intents, errors)
|
||||||
|
- Coordination engine tests (delegation, streaming, multi-agent)
|
||||||
|
- Library-desk client tests (all endpoints with mocked HTTP)
|
||||||
|
- Wiki write operation tests (update, smart-create)
|
||||||
|
- Capability registration tests
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
- Application startup now registers The Librarian with Household Registry
|
||||||
|
- Configuration expanded to support library-desk API integration
|
||||||
|
- **Version loading**: APP_VERSION now dynamically loaded from pyproject.toml
|
||||||
|
|
||||||
|
## [1.0.0a] - 2025-12-11
|
||||||
|
|
||||||
|
### Added
|
||||||
|
- **CI/CD Pipeline**: Release-triggered automated builds
|
||||||
|
- Dockerfile for containerized deployment (Python 3.12-slim, port 8000)
|
||||||
|
- Gitea Actions workflow triggered on release publish
|
||||||
|
- Builds and pushes to git.schweitz.net registry with latest and version tags
|
||||||
|
- Watchtower integration for automatic container updates
|
||||||
|
- **Portainer Stack**: Production deployment configuration
|
||||||
|
- Connects to docker-dataplane network for service discovery
|
||||||
|
- Integration with ollama, searxng, and redis-shared services
|
||||||
|
- Health check endpoint monitoring
|
||||||
|
- Resource limits (1 CPU, 1GB memory)
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
- Version bump to 1.0.0 marking production-ready release
|
||||||
|
|
||||||
## [0.2.5] - 2025-12-07
|
## [0.2.5] - 2025-12-07
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
@@ -297,7 +1021,37 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- CORS middleware
|
- CORS middleware
|
||||||
- Exception handlers (OpenAI-compatible error format)
|
- Exception handlers (OpenAI-compatible error format)
|
||||||
|
|
||||||
[Unreleased]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.2.0...main
|
[Unreleased]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v2.1.0...main
|
||||||
|
[2.1.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v2.0.5...v2.1.0
|
||||||
|
[2.0.5]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v2.0.0...v2.0.5
|
||||||
|
[2.0.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.11.0...v2.0.0
|
||||||
|
[1.11.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.10.0...v1.11.0
|
||||||
|
[1.10.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.9.0...v1.10.0
|
||||||
|
[1.9.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.6...v1.9.0
|
||||||
|
[1.8.6]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.5...v1.8.6
|
||||||
|
[1.8.5]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.4...v1.8.5
|
||||||
|
[1.8.4]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.3...v1.8.4
|
||||||
|
[1.8.3]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.2...v1.8.3
|
||||||
|
[1.8.2]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.1...v1.8.2
|
||||||
|
[1.8.1]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.8.0...v1.8.1
|
||||||
|
[1.8.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.7.0...v1.8.0
|
||||||
|
[1.7.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.6.0...v1.7.0
|
||||||
|
[1.6.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.5.0...v1.6.0
|
||||||
|
[1.5.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.4.0...v1.5.0
|
||||||
|
[1.4.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.3.3...v1.4.0
|
||||||
|
[1.3.3]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.3.2...v1.3.3
|
||||||
|
[1.3.2]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.3.1...v1.3.2
|
||||||
|
[1.3.1]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.3.0...v1.3.1
|
||||||
|
[1.3.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.5...v1.3.0
|
||||||
|
[1.2.5]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.4...v1.2.5
|
||||||
|
[1.2.4]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.3...v1.2.4
|
||||||
|
[1.2.3]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.2...v1.2.3
|
||||||
|
[1.2.2]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.1...v1.2.2
|
||||||
|
[1.2.1]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.2.0...v1.2.1
|
||||||
|
[1.2.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.1.0...v1.2.0
|
||||||
|
[1.1.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v1.0.0a...v1.1.0
|
||||||
|
[1.0.0a]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.2.5...v1.0.0a
|
||||||
|
[0.2.5]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.2.0...v0.2.5
|
||||||
[0.2.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.1.1...v0.2.0
|
[0.2.0]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.1.1...v0.2.0
|
||||||
[0.1.1]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.1.0...v0.1.1
|
[0.1.1]: https://git.schweitz.net/jpmschweitzer/tatlock/compare/v0.1.0...v0.1.1
|
||||||
[0.1.0]: https://git.schweitz.net/jpmschweitzer/tatlock/releases/tag/v0.1.0
|
[0.1.0]: https://git.schweitz.net/jpmschweitzer/tatlock/releases/tag/v0.1.0
|
||||||
|
|||||||
@@ -0,0 +1,196 @@
|
|||||||
|
# CLAUDE.md — tatlock
|
||||||
|
|
||||||
|
Privacy-first homelab butler. An OpenAI-compatible orchestration API over local models, with
|
||||||
|
household staff agents built on PydanticAI. Python 3.12 / FastAPI, `version = "2.4.3"`.
|
||||||
|
Container `tatlock` on `docker-dataplane`, port **8000**. Redis DB **1** (memory), Qdrant for
|
||||||
|
vectors.
|
||||||
|
|
||||||
|
## Ports
|
||||||
|
|
||||||
|
| | Port | How |
|
||||||
|
|---|---|---|
|
||||||
|
| Local dev | **8777** | `make run` — uvicorn reload, logs to `build/logs/server.log` |
|
||||||
|
| Production | **8000** | container; `http://192.168.86.149:8000/health`, external `tatlock.schweitz.net` behind Authentik |
|
||||||
|
|
||||||
|
Test endpoints against `localhost:8777` while developing. `localhost:8000` is the *container*.
|
||||||
|
|
||||||
|
## Live contract
|
||||||
|
|
||||||
|
`http://localhost:8000/openapi.json` — **5 paths**, `title: OpenAI-Compatible API`, `version:
|
||||||
|
2.4.3` (verified 2026-08-09): `/`, `/health`, `/v1/models`, `/v1/chat/completions`,
|
||||||
|
`/v1/responses`. `/v1/responses` is primary; `/v1/chat/completions` exists for Open WebUI.
|
||||||
|
|
||||||
|
**The spec is the public surface, not the system.** The household capability registry is internal
|
||||||
|
and appears nowhere in those 5 paths. Absence from the spec means "not exposed", not "does not
|
||||||
|
exist".
|
||||||
|
|
||||||
|
## Two traps that make the runtime look like the opposite of what it is
|
||||||
|
|
||||||
|
**1. `src/anthropic` loads at startup; `src/ollama` does not — and Ollama is the primary
|
||||||
|
backend.** A cold `import src.main` inside the container shows `agents, anthropic, chat, core,
|
||||||
|
main, models, responses` — no `ollama`. The only import of it is a *function-body* one at
|
||||||
|
`src/anthropic/model_selector.py:230`. Meanwhile `PREFER_CLOUD_BACKEND=false`, so every request
|
||||||
|
actually goes to Ollama and the Claude path is off (see **workspace D-11**). Reading the module list
|
||||||
|
naively gives you exactly the wrong answer: the package that looks live is the disabled fallback,
|
||||||
|
and the one that looks dead is the hot path. Do not conclude anything about backends from
|
||||||
|
`sys.modules`; read the config.
|
||||||
|
|
||||||
|
**2. In-process singletons are empty outside the app.** `get_household_registry()`
|
||||||
|
(`src/core/household_registry.py:334`) in a fresh `docker exec python` returns **0 members**,
|
||||||
|
while the running app serves 2 models from it — it is populated at startup. Import the
|
||||||
|
module-level definitions or ask the endpoint; never import a singleton and assume it is
|
||||||
|
populated.
|
||||||
|
|
||||||
|
## Stack decisions that bind this repo
|
||||||
|
|
||||||
|
Recorded in the workspace vault, not here. Read before assuming anything about the LLM backend:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
/home/jpmschweitzer/.local/bin/pql --vault /mnt/media/Projects decisions read workspace D-11
|
||||||
|
```
|
||||||
|
|
||||||
|
**workspace D-11 — the Claude migration is abandoned. Tatlock stays on Ollama.** Do not resume it and do
|
||||||
|
not treat its remnants as unfinished work. What you will find, and why none of it is a TODO:
|
||||||
|
`ANTHROPIC_MODEL` is set on the container (`claude-sonnet-4-20250514`) and never used because
|
||||||
|
`PREFER_CLOUD_BACKEND=false`; `ANTHROPIC_API_KEY` is a variable reference whose literal was
|
||||||
|
revoked 2026-08-09; `docs/claude-integration.md` documents a capability that exists but is
|
||||||
|
switched off. The cost is deliberate: reasoning stays at `gemma4:e2b` scale because VRAM is
|
||||||
|
shared with Speaches.
|
||||||
|
|
||||||
|
`REDIS_BENCHMARK_DB=6` is allocated on the container but the benchmarking module was never
|
||||||
|
implemented — see the gotcha below. Vestigial, like the Anthropic settings.
|
||||||
|
|
||||||
|
## Critical gotchas
|
||||||
|
|
||||||
|
**ASGITransport does NOT trigger FastAPI lifespan events.** The session-scoped `_initialize_app`
|
||||||
|
fixture in `tests/conftest.py` calls `initialize_application()` explicitly via `asyncio.run()`.
|
||||||
|
Without it the Ollama/Claude health checks never run: `_ollama_available` stays `None` (treated
|
||||||
|
as available, so requests go to Ollama) and `_claude_available` stays `None` (treated as
|
||||||
|
unavailable, so the Claude fallback never engages).
|
||||||
|
|
||||||
|
**AsyncIO scope mismatch.** `asyncio_default_fixture_loop_scope = function` is set in
|
||||||
|
`pyproject.toml`. Session-scoped async fixtures raise `ScopeMismatch`. Use a sync fixture with
|
||||||
|
`asyncio.run()` for session-scoped initialization.
|
||||||
|
|
||||||
|
**The butler persona prompt suppresses local-model tool calling.** With `TATLOCK_SYSTEM_PROMPT`
|
||||||
|
attached, gemma4 reasons about calling the calculator, then answers from memory with wrong
|
||||||
|
arithmetic — a different wrong product each run. `orchestrate_tool_calls()` therefore uses the
|
||||||
|
terse `TATLOCK_ORCHESTRATION_PROMPT`; the persona is applied in `synthesize_from_results()`. Do
|
||||||
|
not reattach the persona prompt to a tool-phase agent. `tool_choice: "required"` via `extra_body`
|
||||||
|
does **not** force Ollama to call tools — advisory at best.
|
||||||
|
|
||||||
|
**Claude Sonnet 5+ rejects sampling parameters.** `temperature`/`top_p`/`top_k` return 400. Use
|
||||||
|
`get_sampling_settings()` from the model selector rather than passing `ModelSettings(temperature=…)`
|
||||||
|
to agents that can run on the Claude fallback. `make test-contracts` pins this.
|
||||||
|
|
||||||
|
**Integration test timeouts** are 120s to match `OLLAMA_TIMEOUT` (300s for the pure-Ollama
|
||||||
|
fallback test, which Claude cannot rescue). GPU-resident numbers measured 2026-08-07 with
|
||||||
|
gemma4:e2b at ~95 tok/s: full Steward → orchestrate → synthesize ~10–13s for simple turns;
|
||||||
|
librarian-routed ~20–25s (not re-measured). **A single turn costs 3 sequential Ollama calls and
|
||||||
|
~710 generated tokens even for "what is 61 plus 12?"** — mostly the model's own reasoning, paid
|
||||||
|
three times. Cold model load is ~36s, avoided while pinned with `keep_alive: -1`; the
|
||||||
|
`OLLAMA_KEEP_ALIVE=2h` default reintroduces it. Older "~35s steward / ~2 min flow" and "11–25s"
|
||||||
|
figures are superseded — do not plan against them. `STEWARD_TIMEOUT` defaults to 60s.
|
||||||
|
|
||||||
|
**`get_benchmark_store` does not exist.** `src/core/benchmarks.py` was never implemented, and
|
||||||
|
`scripts/benchmark_analysis.py` references it and is broken. Do not add mocks for it in tests.
|
||||||
|
|
||||||
|
**Steward tests need the household registry.** Use `register_household_members()` (sync) in
|
||||||
|
fixtures, not `initialize_application()` (async). The steward extracts capabilities from the
|
||||||
|
registry.
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
```bash
|
||||||
|
make setup # venv + all dependencies
|
||||||
|
make run # dev server on 8777, reload, logs to build/logs/server.log
|
||||||
|
make test # unit tests, no external services
|
||||||
|
make test-integration # needs Ollama (and Claude, if enabled)
|
||||||
|
make test-contracts # wire-level contract tests against live service boundaries
|
||||||
|
make lint # ruff linter + formatter check
|
||||||
|
make typecheck # mypy
|
||||||
|
make clean # remove caches and build artifacts
|
||||||
|
```
|
||||||
|
|
||||||
|
Always run pytest through the venv explicitly, to avoid environment mismatch:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
.venv/bin/python -m pytest tests/
|
||||||
|
.venv/bin/python -m pytest tests/core/ -v
|
||||||
|
```
|
||||||
|
|
||||||
|
Dependencies live in `pyproject.toml` (`[project.dependencies]`, `[project.optional-dependencies.dev]`).
|
||||||
|
Copy `.env.example` to `.env` and configure Ollama, Redis and Qdrant hosts.
|
||||||
|
|
||||||
|
**Contract tests before code review.** When the question is "do these two services still agree?",
|
||||||
|
`make test-contracts` answers it by observing the live boundary; reading both codebases only tells
|
||||||
|
you what should happen. Semantics: unreachable → skip, reachable-but-wrong-shape → fail.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
Domain-first under `src/`: `agents/` (steward, librarian, biographer, housekeeper, tatlock_core),
|
||||||
|
`core/`, `chat/`, `responses/`, `models/`, `ollama/`, `anthropic/`. Two tiers — the Steward routes,
|
||||||
|
Tatlock coordinates. Group new work by domain, not by file type.
|
||||||
|
|
||||||
|
## Internal service access
|
||||||
|
|
||||||
|
`http://localhost:3002` reaches Gitea directly, bypassing Authentik SSO — verified returning
|
||||||
|
`{"version":"1.27.1"}`. Useful for reading a sibling repo's raw files:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://localhost:3002/jpmschweitzer/library-desk/raw/branch/main/README.md
|
||||||
|
```
|
||||||
|
|
||||||
|
The old AGENTS.md pointed at **`portainer-core`** for full-stack documentation. That repo is
|
||||||
|
**deprecated** and must not be used as a source of infra facts; it was merged into
|
||||||
|
`system-admin-toj/containers/`, where `CONTAINERS.md` is the live inventory.
|
||||||
|
|
||||||
|
## Work tracking
|
||||||
|
|
||||||
|
Work lives in **pql**, not a markdown TODO. **This repo's vault is standalone** — its tickets and
|
||||||
|
internal decisions live here in `.pql/` and `governance/`, and travel with a clone, because
|
||||||
|
`.pql/changelog/` is committed and replayed by the git hooks (workspace D-15). The databases are gitignored
|
||||||
|
and rebuildable with `pql plan rebuild`.
|
||||||
|
|
||||||
|
`pql` is **not** on the non-interactive `PATH` — invoke it as `/home/jpmschweitzer/.local/bin/pql`.
|
||||||
|
From inside this repo no `--vault` is needed; pql anchors at the nearest `.git/` ancestor.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
/home/jpmschweitzer/.local/bin/pql ticket list # this repo's open work
|
||||||
|
/home/jpmschweitzer/.local/bin/pql plan whatsnext # next unblocked item, with context
|
||||||
|
/home/jpmschweitzer/.local/bin/pql decisions list # this repo's own decisions
|
||||||
|
```
|
||||||
|
|
||||||
|
Stack decisions that constrain this service need the flag:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
/home/jpmschweitzer/.local/bin/pql --vault /mnt/media/Projects decisions list --domain tatlock-api
|
||||||
|
```
|
||||||
|
|
||||||
|
The workspace domain is `tatlock-api`, not `tatlock` — pql rejects a domain stem that prefixes
|
||||||
|
another, and `tatlock` prefixes `tatlock-ui`. A `tatlock-api -> tatlock` symlink at the workspace
|
||||||
|
root makes the directory answer to both (workspace D-15).
|
||||||
|
|
||||||
|
Note `ticket new --decision D-N` resolves ids within **one** vault, so a ticket here cannot link
|
||||||
|
to a workspace decision. Cite the id in the ticket body instead.
|
||||||
|
|
||||||
|
## Git
|
||||||
|
|
||||||
|
- **History is linear — no merge commits.** Work on `main`, or a short-lived branch that is
|
||||||
|
fast-forwarded and deleted. This repo's AGENTS.md mandated a feature branch for every change;
|
||||||
|
that rule was retired workspace-wide on 2026-08-08 and does not apply.
|
||||||
|
- **Conventional Commits**: `feat:`, `fix:`, `refactor:`, `docs:`, `chore:`.
|
||||||
|
- **Stage explicitly. Never `git add -A`** — denied by policy, and it sweeps in whatever else is
|
||||||
|
dirty, including secrets.
|
||||||
|
- Update `CHANGELOG.md` for every user-facing change, under `[Unreleased]`.
|
||||||
|
|
||||||
|
## Releasing
|
||||||
|
|
||||||
|
Test locally first — the build-deploy loop is slow. Deploy only when a feature is complete.
|
||||||
|
|
||||||
|
1. Ask whether a deploy is wanted; it is not automatic.
|
||||||
|
2. Bump `version` in `pyproject.toml` (patch for fixes, minor for features).
|
||||||
|
3. Move `[Unreleased]` entries into a dated section in `CHANGELOG.md`.
|
||||||
|
4. Stage the changed files by name, commit, tag `vX.Y.Z`, `git push origin main --tags`.
|
||||||
|
5. Gitea CI builds and pushes on the tag; Watchtower deploys.
|
||||||
|
6. Verify: `curl http://192.168.86.149:8000/health`.
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
FROM python:3.12-slim
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
RUN apt-get update && apt-get install -y curl \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
COPY pyproject.toml ./
|
||||||
|
RUN pip install --no-cache-dir .
|
||||||
|
|
||||||
|
COPY src/ ./src/
|
||||||
|
|
||||||
|
ENV PYTHONPATH=/app
|
||||||
|
|
||||||
|
EXPOSE 8000
|
||||||
|
|
||||||
|
CMD ["uvicorn", "src.main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]
|
||||||
@@ -1,883 +0,0 @@
|
|||||||
# Tatlock Implementation Roadmap
|
|
||||||
|
|
||||||
> **Reference**: See [PHILOSOPHY.md](PHILOSOPHY.md) for the target architecture and vision
|
|
||||||
|
|
||||||
This document outlines the phased implementation plan to transform the current OpenAI-compatible API into the full Tatlock household butler system.
|
|
||||||
|
|
||||||
## Current State (v0.1.1+ - Phase 1 Mostly Complete)
|
|
||||||
|
|
||||||
**What we have**:
|
|
||||||
- ✅ **The Orchestrator** - FastAPI infrastructure layer
|
|
||||||
- OpenAI-compatible API endpoints (Responses API + Chat Completions)
|
|
||||||
- Streaming coordination and conversation management
|
|
||||||
- Response format with reasoning support
|
|
||||||
- Test infrastructure (131 tests, 81.78% coverage)
|
|
||||||
- ✅ **Tatlock Agent** - Real PydanticAI integration
|
|
||||||
- Connected to Ollama (mistral-nemo:latest)
|
|
||||||
- British butler personality with research mindset
|
|
||||||
- Streaming responses with reasoning
|
|
||||||
- Tool calling framework functional
|
|
||||||
- ✅ **Permanent Tools**
|
|
||||||
- Calculator (safe mathematical expressions)
|
|
||||||
- Date/Time toolkit (current time, relative dates, time differences)
|
|
||||||
- Web search (SearXNG integration)
|
|
||||||
- ✅ Mock agent (lorem-tester for testing)
|
|
||||||
- ✅ Agent interface abstraction
|
|
||||||
|
|
||||||
**What we need**:
|
|
||||||
- **The Household** - Full multi-agent coordination:
|
|
||||||
- The Steward (first-tier request analysis)
|
|
||||||
- Tatlock coordination layer (expert agent delegation)
|
|
||||||
- Expert household staff agents (Librarian, Developer, Handyman, etc.)
|
|
||||||
- Multi-tenant database architecture
|
|
||||||
- Containerized service ecosystem
|
|
||||||
- MCP (Model Context Protocol) integration
|
|
||||||
- Dynamic model switching for specialized tasks
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 1: Real LLM Integration - PydanticAI + Tools
|
|
||||||
|
|
||||||
**Goal**: Connect to actual language models and establish the base plumbing
|
|
||||||
|
|
||||||
**Note**: Ollama is an external service dependency (already running separately)
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **PydanticAI Integration** ✅
|
|
||||||
- PydanticAI → Ollama connection ✅
|
|
||||||
- Agent creation patterns ✅
|
|
||||||
- Streaming response handling ✅
|
|
||||||
- Error handling and retries ✅
|
|
||||||
|
|
||||||
2. **Convert Tatlock Agent** ✅
|
|
||||||
- Convert Tatlock agent from mock to PydanticAI ✅
|
|
||||||
- British butler personality prompt ✅
|
|
||||||
- Research-oriented mindset ✅
|
|
||||||
- Streaming to reasoning output ✅
|
|
||||||
- Tool calling framework setup ✅
|
|
||||||
|
|
||||||
3. **Permanent Tools** ✅
|
|
||||||
- Calculator: Safe mathematical expression evaluation ✅
|
|
||||||
- Date/Time toolkit: Current time, relative dates, time differences ✅
|
|
||||||
- Web search: SearXNG integration (external service) ✅
|
|
||||||
- Tool registration with PydanticAI ✅
|
|
||||||
|
|
||||||
4. **Testing Infrastructure** ✅
|
|
||||||
- Integration tests with real LLM ✅
|
|
||||||
- Tool functionality tests ✅
|
|
||||||
- Response quality validation ✅
|
|
||||||
- 131 tests, 81.78% coverage ✅
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [x] **PydanticAI agents can call Ollama** (mistral-nemo:latest)
|
|
||||||
- [x] **Streaming works end-to-end**
|
|
||||||
- [x] **Tool calling framework functional**
|
|
||||||
- [x] **Permanent tools working** (calculator, date/time, search)
|
|
||||||
- [x] **Tests pass with real LLM**
|
|
||||||
- [ ] Can switch models dynamically (e.g., Codestral for code)
|
|
||||||
|
|
||||||
### Status
|
|
||||||
**✅ MOSTLY COMPLETE** - Tatlock agent functional with permanent tools
|
|
||||||
|
|
||||||
### Remaining Work
|
|
||||||
- Dynamic model switching for specialized tasks (e.g., Codestral for coding)
|
|
||||||
|
|
||||||
### Why First?
|
|
||||||
Without real LLM integration, we can't meaningfully implement the Steward/Butler pattern. Everything else depends on having actual AI agents working.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 2: Orchestration Layer - The Steward
|
|
||||||
|
|
||||||
**Goal**: Implement the first-tier LLM call for tool/agent selection
|
|
||||||
|
|
||||||
**Purpose**: The Steward performs crucial preparatory work before Tatlock engages with a request. By analyzing incoming requests and determining which tools, services, and household staff members will be needed, the Steward creates a curated recommendation that streamlines Tatlock's work and prevents cognitive overload.
|
|
||||||
|
|
||||||
### Core Architecture
|
|
||||||
|
|
||||||
The Steward operates as the first tier in the two-tier request flow:
|
|
||||||
|
|
||||||
```
|
|
||||||
User Request → Orchestrator → Steward Analysis → Recommendations → Tatlock (with scoped tools/agents)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Key Principle**: The Steward narrows the scope to only relevant capabilities, making Tatlock's decision-making cleaner and more focused.
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
#### 1. Tool & Agent Registry System
|
|
||||||
|
|
||||||
**Purpose**: Centralized catalog of all available capabilities for the Steward to recommend
|
|
||||||
|
|
||||||
**Implementation Details**:
|
|
||||||
- **Registry Module** (`src/core/registry.py`)
|
|
||||||
- Tool registration decorator pattern
|
|
||||||
- Agent registration with capability metadata
|
|
||||||
- Category-based organization (computation, information, automation, communication)
|
|
||||||
- Dynamic tool/agent discovery and loading
|
|
||||||
|
|
||||||
- **Tool Metadata Schema**
|
|
||||||
```python
|
|
||||||
{
|
|
||||||
"name": "calculator",
|
|
||||||
"category": "computation",
|
|
||||||
"description": "Safe mathematical expression evaluation",
|
|
||||||
"capabilities": ["arithmetic", "algebra", "trigonometry"],
|
|
||||||
"cost": "low", # computational cost indicator
|
|
||||||
"requires_network": false
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
- **Agent Metadata Schema**
|
|
||||||
```python
|
|
||||||
{
|
|
||||||
"name": "developer",
|
|
||||||
"role": "The Developer",
|
|
||||||
"category": "technical",
|
|
||||||
"description": "Software development assistance",
|
|
||||||
"domains": ["code_generation", "debugging", "architecture"],
|
|
||||||
"specialized_model": "codestral", # optional
|
|
||||||
"cost": "high"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
- **Registry API**
|
|
||||||
- `get_all_tools()` - List all available tools
|
|
||||||
- `get_all_agents()` - List all expert agents
|
|
||||||
- `get_by_category(category)` - Filter by category
|
|
||||||
- `search_by_capability(query)` - Semantic search (future: vector search)
|
|
||||||
|
|
||||||
**Testing**:
|
|
||||||
- Unit tests for registration and retrieval
|
|
||||||
- Test dynamic loading of new tools/agents
|
|
||||||
- Validate metadata schemas
|
|
||||||
|
|
||||||
#### 2. Steward PydanticAI Agent
|
|
||||||
|
|
||||||
**Purpose**: First-tier LLM that analyzes requests and recommends relevant tools/agents
|
|
||||||
|
|
||||||
**Implementation Details**:
|
|
||||||
|
|
||||||
- **Agent Module** (`src/agents/steward.py`)
|
|
||||||
```python
|
|
||||||
from pydantic_ai import Agent, RunContext
|
|
||||||
from pydantic import BaseModel
|
|
||||||
|
|
||||||
class StewardRecommendation(BaseModel):
|
|
||||||
"""Structured output from Steward analysis"""
|
|
||||||
recommended_tools: list[str]
|
|
||||||
recommended_agents: list[str]
|
|
||||||
reasoning: str
|
|
||||||
estimated_complexity: str # "simple", "moderate", "complex"
|
|
||||||
requires_multi_step: bool
|
|
||||||
|
|
||||||
steward = Agent(
|
|
||||||
'ollama:mistral-nemo', # Same base model as Tatlock
|
|
||||||
result_type=StewardRecommendation,
|
|
||||||
system_prompt="""..."""
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
- **System Prompt Engineering**
|
|
||||||
- Role: Estate steward responsible for efficient household coordination
|
|
||||||
- Task: Analyze requests to determine needed resources
|
|
||||||
- Output: Structured recommendations with reasoning
|
|
||||||
- Constraints: Be conservative (recommend only truly relevant capabilities)
|
|
||||||
- Context: Full registry of available tools and agents
|
|
||||||
|
|
||||||
- **Steward Tools**
|
|
||||||
```python
|
|
||||||
@steward.tool
|
|
||||||
def get_available_capabilities(ctx: RunContext) -> dict:
|
|
||||||
"""Get catalog of all available tools and agents."""
|
|
||||||
return {
|
|
||||||
"tools": registry.get_all_tools(),
|
|
||||||
"agents": registry.get_all_agents()
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
- **Request Analysis Flow**
|
|
||||||
1. Receive user request
|
|
||||||
2. Query capability registry via tool
|
|
||||||
3. Analyze request for required capabilities
|
|
||||||
4. Generate structured recommendation
|
|
||||||
5. Format as note to Tatlock
|
|
||||||
|
|
||||||
**Testing**:
|
|
||||||
- Test various request types (simple, complex, multi-domain)
|
|
||||||
- Verify recommendations are relevant and not over-inclusive
|
|
||||||
- Test structured output parsing
|
|
||||||
- Validate reasoning quality
|
|
||||||
|
|
||||||
#### 3. Request Preprocessing Pipeline
|
|
||||||
|
|
||||||
**Purpose**: Integration layer that routes requests through Steward before Tatlock
|
|
||||||
|
|
||||||
**Implementation Details**:
|
|
||||||
|
|
||||||
- **Preprocessing Module** (`src/core/preprocessing.py`)
|
|
||||||
```python
|
|
||||||
async def preprocess_request(user_request: str) -> EnrichedRequest:
|
|
||||||
"""
|
|
||||||
1. Call Steward for analysis
|
|
||||||
2. Get recommendations
|
|
||||||
3. Enrich original request
|
|
||||||
4. Return scoped context for Tatlock
|
|
||||||
"""
|
|
||||||
# Get Steward analysis
|
|
||||||
steward_result = await steward.run(user_request)
|
|
||||||
recommendations = steward_result.data
|
|
||||||
|
|
||||||
# Create note to Tatlock
|
|
||||||
steward_note = format_steward_note(recommendations)
|
|
||||||
|
|
||||||
# Build scoped tool/agent list
|
|
||||||
scoped_tools = get_scoped_tools(recommendations.recommended_tools)
|
|
||||||
scoped_agents = get_scoped_agents(recommendations.recommended_agents)
|
|
||||||
|
|
||||||
return EnrichedRequest(
|
|
||||||
original_request=user_request,
|
|
||||||
steward_note=steward_note,
|
|
||||||
available_tools=scoped_tools,
|
|
||||||
available_agents=scoped_agents,
|
|
||||||
metadata=recommendations
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
- **Note Formatting**
|
|
||||||
```
|
|
||||||
=== Internal Note from the Steward ===
|
|
||||||
|
|
||||||
Request Analysis:
|
|
||||||
{steward reasoning}
|
|
||||||
|
|
||||||
Recommended Tools:
|
|
||||||
- calculator: For mathematical computations
|
|
||||||
- web_search: To find current information
|
|
||||||
|
|
||||||
Recommended Household Staff:
|
|
||||||
- The Developer: For code generation assistance
|
|
||||||
|
|
||||||
Estimated Complexity: moderate
|
|
||||||
===================================
|
|
||||||
|
|
||||||
[Original User Request]
|
|
||||||
```
|
|
||||||
|
|
||||||
- **Orchestrator Integration**
|
|
||||||
- Modify `src/responses/service.py` to call preprocessing
|
|
||||||
- Prepend Steward note to request before sending to Tatlock
|
|
||||||
- Limit Tatlock's tool access to recommended tools only
|
|
||||||
- Stream Steward's reasoning to output
|
|
||||||
|
|
||||||
**Testing**:
|
|
||||||
- Integration tests for full preprocessing flow
|
|
||||||
- Test request enrichment format
|
|
||||||
- Verify tool scoping works correctly
|
|
||||||
- Test streaming of Steward reasoning
|
|
||||||
|
|
||||||
#### 4. Real-Time Transparency
|
|
||||||
|
|
||||||
**Purpose**: Stream Steward's analysis to user's reasoning output
|
|
||||||
|
|
||||||
**Implementation Details**:
|
|
||||||
|
|
||||||
- **Streaming Integration** (`src/responses/streaming.py`)
|
|
||||||
- Add Steward analysis phase to stream
|
|
||||||
- Format as reasoning item
|
|
||||||
- Include recommendation summary
|
|
||||||
|
|
||||||
- **Example Output to User**:
|
|
||||||
```
|
|
||||||
[Reasoning]
|
|
||||||
Consulting the Steward for resource planning...
|
|
||||||
|
|
||||||
The Steward's Analysis:
|
|
||||||
- Request requires mathematical computation
|
|
||||||
- Need to verify current information via web search
|
|
||||||
- May benefit from Developer's code expertise
|
|
||||||
|
|
||||||
Recommended: calculator, web_search, The Developer
|
|
||||||
|
|
||||||
Proceeding with scoped resources...
|
|
||||||
```
|
|
||||||
|
|
||||||
**Testing**:
|
|
||||||
- Test streaming of Steward analysis
|
|
||||||
- Verify formatting in Open WebUI
|
|
||||||
- Test error handling if Steward fails
|
|
||||||
|
|
||||||
#### 5. Model Efficiency Optimization
|
|
||||||
|
|
||||||
**Purpose**: Ensure the base model stays loaded in VRAM
|
|
||||||
|
|
||||||
**Implementation Details**:
|
|
||||||
|
|
||||||
- **Shared Model Configuration**
|
|
||||||
- Both Steward and Tatlock use `ollama:mistral-nemo` by default
|
|
||||||
- Sequential calls (Steward → Tatlock) keep model hot
|
|
||||||
- No reload delays between tiers
|
|
||||||
|
|
||||||
- **Performance Monitoring**
|
|
||||||
- Log response times for Steward calls
|
|
||||||
- Track total request latency (Steward + Tatlock)
|
|
||||||
- Identify optimization opportunities
|
|
||||||
|
|
||||||
**Testing**:
|
|
||||||
- Benchmark Steward → Tatlock call latency
|
|
||||||
- Verify model stays loaded between calls
|
|
||||||
- Test performance under load
|
|
||||||
|
|
||||||
### Implementation Strategy
|
|
||||||
|
|
||||||
#### Week 1-2: Foundation
|
|
||||||
- [ ] Design and implement registry system
|
|
||||||
- [ ] Create tool/agent metadata schemas
|
|
||||||
- [ ] Build registry API with tests
|
|
||||||
- [ ] Migrate existing tools to registry
|
|
||||||
|
|
||||||
#### Week 3-4: Steward Agent
|
|
||||||
- [ ] Create Steward PydanticAI agent
|
|
||||||
- [ ] Engineer system prompt for analysis
|
|
||||||
- [ ] Implement structured recommendation output
|
|
||||||
- [ ] Add registry query tool
|
|
||||||
- [ ] Test with various request types
|
|
||||||
|
|
||||||
#### Week 5-6: Integration
|
|
||||||
- [ ] Build request preprocessing pipeline
|
|
||||||
- [ ] Implement note formatting
|
|
||||||
- [ ] Integrate with Orchestrator
|
|
||||||
- [ ] Add streaming transparency
|
|
||||||
- [ ] Tool scoping for Tatlock
|
|
||||||
|
|
||||||
#### Week 7: Testing & Refinement
|
|
||||||
- [ ] End-to-end integration tests
|
|
||||||
- [ ] Performance optimization
|
|
||||||
- [ ] Prompt refinement based on results
|
|
||||||
- [ ] Documentation and examples
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
|
|
||||||
- [ ] **Steward analyzes incoming requests** using PydanticAI agent
|
|
||||||
- [ ] **Produces structured recommendations** (tools, agents, reasoning)
|
|
||||||
- [ ] **Recommendations formatted as prepended note** to Tatlock
|
|
||||||
- [ ] **Tool registry is queryable and extensible** via clean API
|
|
||||||
- [ ] **Steward output visible in reasoning stream** for transparency
|
|
||||||
- [ ] **Only recommended tools available** to Tatlock (scoped context)
|
|
||||||
- [ ] **Base model stays loaded** between Steward and Tatlock calls
|
|
||||||
- [ ] **Recommendations are accurate** (not over/under-inclusive)
|
|
||||||
- [ ] **Integration tests pass** for full Steward → Tatlock flow
|
|
||||||
|
|
||||||
### Performance Targets
|
|
||||||
|
|
||||||
- **Steward Analysis Time**: < 2 seconds for typical requests
|
|
||||||
- **Total Added Latency**: < 3 seconds including streaming
|
|
||||||
- **Recommendation Accuracy**: > 90% relevance (manual evaluation)
|
|
||||||
- **Model Reload Delay**: 0 seconds (model stays hot)
|
|
||||||
|
|
||||||
### Risk Mitigation
|
|
||||||
|
|
||||||
**Risk**: Steward recommendations too broad (defeats purpose)
|
|
||||||
- Mitigation: Conservative prompt engineering, test with diverse requests, iterate
|
|
||||||
|
|
||||||
**Risk**: Added latency unacceptable to users
|
|
||||||
- Mitigation: Stream Steward reasoning for transparency, optimize prompt, parallel processing where possible
|
|
||||||
|
|
||||||
**Risk**: Tool registry becomes unwieldy
|
|
||||||
- Mitigation: Good categorization, semantic search (future), regular pruning
|
|
||||||
|
|
||||||
**Risk**: Steward and Tatlock models compete for VRAM
|
|
||||||
- Mitigation: Use same base model, sequential calls, monitor memory
|
|
||||||
|
|
||||||
### Future Enhancements (Post-Phase 2)
|
|
||||||
|
|
||||||
- **Semantic Search**: Vector-based capability search instead of metadata lookup
|
|
||||||
- **Learning from Usage**: Track which recommendations work well, adjust over time
|
|
||||||
- **Confidence Scores**: Steward provides confidence for each recommendation
|
|
||||||
- **Request Classification**: Cache classifications for similar requests
|
|
||||||
- **Multi-Model Support**: Allow Steward to recommend specialized models for specific tasks
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
|
|
||||||
**7-8 weeks** - Core intelligence routing with comprehensive implementation
|
|
||||||
|
|
||||||
### Why Second?
|
|
||||||
|
|
||||||
The Steward is the foundation of the household architecture. Without it, we'd need to expose all tools/agents to Tatlock, creating cognitive overload and poor decision-making. The Steward enables the focused expertise pattern that makes the whole system work.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 3: The Butler - Tatlock Agent
|
|
||||||
|
|
||||||
**Goal**: Implement the second-tier coordinator with personality within the existing Orchestrator infrastructure
|
|
||||||
|
|
||||||
**Context**: The Orchestrator (FastAPI infrastructure) already exists. This phase implements the real Tatlock PydanticAI agent to replace the current mock agent.
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **Butler Agent (Tatlock)**
|
|
||||||
- PydanticAI agent implementation within Orchestrator
|
|
||||||
- Personality prompt engineering (witty British butler)
|
|
||||||
- Tool calling framework
|
|
||||||
- Multi-agent coordination logic
|
|
||||||
|
|
||||||
2. **Scoped Tool Access**
|
|
||||||
- Filter tools based on Steward recommendations
|
|
||||||
- Dynamic tool loading for Butler context
|
|
||||||
- Tool execution framework
|
|
||||||
- Result aggregation
|
|
||||||
|
|
||||||
3. **Real-Time Reasoning Output**
|
|
||||||
- Stream all Butler activities to reasoning output
|
|
||||||
- Tool call progress indicators
|
|
||||||
- Expert agent consultation messages
|
|
||||||
- Wait time transparency
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] Tatlock receives enriched requests (user + Steward notes)
|
|
||||||
- [ ] Only recommended tools are available
|
|
||||||
- [ ] Tatlock coordinates multiple tool calls
|
|
||||||
- [ ] All actions streamed to reasoning output
|
|
||||||
- [ ] Responses have consistent personality
|
|
||||||
- [ ] Synthesizes multi-source results coherently
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**4-5 weeks** - Complex coordination logic
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 4: Expert Household Staff - Core Agents
|
|
||||||
|
|
||||||
**Goal**: Implement the initial set of domain-specific expert agents
|
|
||||||
|
|
||||||
### Priority Expert Agents
|
|
||||||
|
|
||||||
1. **The Librarian** (Research & Knowledge Management) ⭐ **Priority**
|
|
||||||
- Research assistance and synthesis
|
|
||||||
- Automatic research dossier generation
|
|
||||||
- Knowledge base queries and organization
|
|
||||||
- Reference management
|
|
||||||
- Wiki integration (future: dedicated wiki container)
|
|
||||||
- Mind map maintenance (future)
|
|
||||||
- *Rationale: Helps guide development priorities through better research*
|
|
||||||
|
|
||||||
2. **The Developer** (Software Development)
|
|
||||||
- Code generation assistance
|
|
||||||
- Debugging support
|
|
||||||
- Documentation generation
|
|
||||||
- Architecture guidance
|
|
||||||
- *Rationale: Directly supports building the system itself*
|
|
||||||
|
|
||||||
3. **The Handyman** (System Maintenance)
|
|
||||||
- System status queries
|
|
||||||
- Log analysis
|
|
||||||
- Basic troubleshooting
|
|
||||||
- Infrastructure monitoring
|
|
||||||
|
|
||||||
4. **The Secretary** (Scheduling & Organization)
|
|
||||||
- Calendar integration (placeholder)
|
|
||||||
- Task management (placeholder)
|
|
||||||
- Reminder system
|
|
||||||
- Schedule conflict detection
|
|
||||||
|
|
||||||
5. **The Housekeeper** (Home Automation)
|
|
||||||
- Device control interface
|
|
||||||
- Status queries
|
|
||||||
- Automation triggers
|
|
||||||
- Environmental monitoring
|
|
||||||
|
|
||||||
### Each Agent Includes
|
|
||||||
- Specialized prompt and personality
|
|
||||||
- Domain-specific tools
|
|
||||||
- MCP integration points (where applicable)
|
|
||||||
- Integration with Butler orchestration
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] Each agent implemented as separate module
|
|
||||||
- [ ] Agents callable via tool framework
|
|
||||||
- [ ] Agents use specialized prompts
|
|
||||||
- [ ] Results integrate cleanly with Butler
|
|
||||||
- [ ] Can invoke specialized models (e.g., Codestral for Developer)
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**6-8 weeks** - Parallel development possible
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 5: Persistence Layer - Database & Multi-Tenancy
|
|
||||||
|
|
||||||
**Goal**: Add persistent storage and multi-user support when needed
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **PostgreSQL Integration**
|
|
||||||
- Docker compose configuration for PostgreSQL
|
|
||||||
- Database schema design with tenant isolation
|
|
||||||
- Alembic migrations setup
|
|
||||||
- SQLAlchemy models
|
|
||||||
|
|
||||||
2. **Multi-Tenant Architecture**
|
|
||||||
- Tenant identification middleware
|
|
||||||
- Tenant-scoped database sessions
|
|
||||||
- User authentication system (basic)
|
|
||||||
- Per-tenant data isolation
|
|
||||||
|
|
||||||
3. **Core Data Models**
|
|
||||||
- Users and tenants
|
|
||||||
- Conversations and messages (migrate from in-memory)
|
|
||||||
- Agent interactions log
|
|
||||||
- System configuration and preferences
|
|
||||||
|
|
||||||
4. **Migration Strategy**
|
|
||||||
- Gradual migration from in-memory to database
|
|
||||||
- Backward compatibility during transition
|
|
||||||
- Data export/import utilities
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] PostgreSQL container running
|
|
||||||
- [ ] Multiple users can authenticate separately
|
|
||||||
- [ ] Each user sees only their own data
|
|
||||||
- [ ] Conversations persist across restarts
|
|
||||||
- [ ] Database migrations work correctly
|
|
||||||
- [ ] Tests verify tenant isolation
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**3-4 weeks** - Data layer foundation
|
|
||||||
|
|
||||||
### Why Later?
|
|
||||||
The core orchestration (Steward → Butler → Experts) can work entirely with in-memory state. We only need database persistence when we want conversations to survive restarts and multiple users to have isolated experiences.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 6: Extended Services Integration
|
|
||||||
|
|
||||||
**Goal**: Connect to additional supporting services
|
|
||||||
|
|
||||||
### Services to Integrate
|
|
||||||
|
|
||||||
1. **Redis (Memory & Caching)**
|
|
||||||
- Docker compose setup
|
|
||||||
- Conversation cache
|
|
||||||
- Short-term memory
|
|
||||||
- Session management
|
|
||||||
|
|
||||||
3. **Qdrant (Vector Storage)**
|
|
||||||
- Docker compose setup
|
|
||||||
- Long-term memory embeddings
|
|
||||||
- Semantic search
|
|
||||||
- Conversation history vectors
|
|
||||||
|
|
||||||
4. **SearxNG (Web Search)**
|
|
||||||
- Docker compose setup
|
|
||||||
- Search tool integration
|
|
||||||
- Result processing
|
|
||||||
- Privacy-preserving queries
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] All services defined in docker-compose.yml
|
|
||||||
- [ ] Services communicate correctly
|
|
||||||
- [ ] Tatlock can invoke web search
|
|
||||||
- [ ] Redis used for session data
|
|
||||||
- [ ] Qdrant stores conversation embeddings
|
|
||||||
- [ ] Ollama serves the base model
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**3-4 weeks** - Infrastructure setup
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 7: MCP (Model Context Protocol) Integration
|
|
||||||
|
|
||||||
**Goal**: Enable rich tool integrations via MCP
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **MCP Server Framework**
|
|
||||||
- MCP server implementation
|
|
||||||
- Tool registration via MCP
|
|
||||||
- Schema validation
|
|
||||||
- Error handling
|
|
||||||
|
|
||||||
2. **MCP Client in Agents**
|
|
||||||
- PydanticAI MCP integration
|
|
||||||
- Tool discovery from MCP servers
|
|
||||||
- Dynamic tool loading
|
|
||||||
- Result processing
|
|
||||||
|
|
||||||
3. **Initial MCP Tools**
|
|
||||||
- File system operations
|
|
||||||
- Database queries
|
|
||||||
- API integrations
|
|
||||||
- System commands
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] MCP server running
|
|
||||||
- [ ] Tools exposed via MCP protocol
|
|
||||||
- [ ] Agents can discover and use MCP tools
|
|
||||||
- [ ] New tools addable without code changes
|
|
||||||
- [ ] MCP tools visible in Steward recommendations
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**3-4 weeks** - Standards-based integration
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 8: Advanced Memory & Context
|
|
||||||
|
|
||||||
**Goal**: Implement sophisticated memory and context management
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **Long-Term Memory**
|
|
||||||
- Conversation embedding pipeline
|
|
||||||
- Semantic search over history
|
|
||||||
- Memory consolidation
|
|
||||||
- Relevance ranking
|
|
||||||
|
|
||||||
2. **Context Management**
|
|
||||||
- Smart context window trimming
|
|
||||||
- Conversation branching
|
|
||||||
- Topic tracking
|
|
||||||
- Memory retrieval integration
|
|
||||||
|
|
||||||
3. **Personalization**
|
|
||||||
- User preference learning
|
|
||||||
- Interaction pattern analysis
|
|
||||||
- Adaptive responses
|
|
||||||
- Custom agent personalities per user
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] Conversations automatically embedded to Qdrant
|
|
||||||
- [ ] Relevant history retrieved for new requests
|
|
||||||
- [ ] Context stays within model limits
|
|
||||||
- [ ] User preferences affect responses
|
|
||||||
- [ ] Memory improves over time
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**4-5 weeks** - AI/ML heavy
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 9: Extended Household Staff
|
|
||||||
|
|
||||||
**Goal**: Add specialized agents for additional domains
|
|
||||||
|
|
||||||
### Future Agents
|
|
||||||
|
|
||||||
1. **The Librarian** (Knowledge Management)
|
|
||||||
- Personal documentation indexing
|
|
||||||
- Research assistance
|
|
||||||
- Knowledge base queries
|
|
||||||
- Reference management
|
|
||||||
|
|
||||||
2. **The Accountant** (Financial Tracking)
|
|
||||||
- Expense tracking
|
|
||||||
- Budget monitoring
|
|
||||||
- Financial reports
|
|
||||||
- Transaction categorization
|
|
||||||
|
|
||||||
3. **The Chef** (Meal Planning)
|
|
||||||
- Recipe management
|
|
||||||
- Meal planning
|
|
||||||
- Nutrition tracking
|
|
||||||
- Grocery lists
|
|
||||||
|
|
||||||
4. **Others as Needed**
|
|
||||||
- Domain-specific as requirements emerge
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] Each new agent follows household pattern
|
|
||||||
- [ ] Integrates with Steward/Butler flow
|
|
||||||
- [ ] Has appropriate specialized tools
|
|
||||||
- [ ] Documented in PHILOSOPHY.md updates
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**Ongoing** - Add as needed
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 10: User Experience Refinement
|
|
||||||
|
|
||||||
**Goal**: Polish the interaction experience
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **Personality Tuning**
|
|
||||||
- Refine Tatlock's wit and tone
|
|
||||||
- Consistent household character
|
|
||||||
- Cultural references appropriate
|
|
||||||
- Humor that doesn't annoy
|
|
||||||
|
|
||||||
2. **Transparency Improvements**
|
|
||||||
- Better progress indicators
|
|
||||||
- Clearer reasoning explanations
|
|
||||||
- Informative wait messages
|
|
||||||
- Error message clarity
|
|
||||||
|
|
||||||
3. **Performance Optimization**
|
|
||||||
- Response time improvements
|
|
||||||
- Model loading optimization
|
|
||||||
- Caching strategies
|
|
||||||
- Streaming smoothness
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] Users find Tatlock engaging
|
|
||||||
- [ ] Wait times feel reasonable
|
|
||||||
- [ ] Errors are understandable
|
|
||||||
- [ ] System feels responsive
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**Ongoing** - Continuous improvement
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Phase 11: Production Hardening
|
|
||||||
|
|
||||||
**Goal**: Make the system production-ready for homelab deployment
|
|
||||||
|
|
||||||
### Deliverables
|
|
||||||
|
|
||||||
1. **Deployment**
|
|
||||||
- Complete docker-compose stack
|
|
||||||
- Environment configuration
|
|
||||||
- Backup strategies
|
|
||||||
- Update procedures
|
|
||||||
|
|
||||||
2. **Monitoring**
|
|
||||||
- Health checks
|
|
||||||
- Performance metrics
|
|
||||||
- Error tracking
|
|
||||||
- Usage analytics
|
|
||||||
|
|
||||||
3. **Security**
|
|
||||||
- Authentication hardening
|
|
||||||
- Rate limiting
|
|
||||||
- Input validation
|
|
||||||
- Audit logging
|
|
||||||
|
|
||||||
4. **Documentation**
|
|
||||||
- Installation guide
|
|
||||||
- Configuration reference
|
|
||||||
- Troubleshooting guide
|
|
||||||
- Architecture documentation
|
|
||||||
|
|
||||||
### Success Criteria
|
|
||||||
- [ ] One-command deployment
|
|
||||||
- [ ] System health is monitorable
|
|
||||||
- [ ] Secure for homelab use
|
|
||||||
- [ ] Well documented
|
|
||||||
|
|
||||||
### Estimated Effort
|
|
||||||
**3-4 weeks** - Production polish
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Dependencies Between Phases
|
|
||||||
|
|
||||||
```
|
|
||||||
Phase 1 (Ollama + PydanticAI) ← Foundation for all AI
|
|
||||||
↓
|
|
||||||
Phase 2 (Steward)
|
|
||||||
↓
|
|
||||||
Phase 3 (Butler/Tatlock)
|
|
||||||
↓
|
|
||||||
Phase 4 (Expert Agents) ← Phase 7 (MCP) can enhance
|
|
||||||
↓
|
|
||||||
Phase 5 (Database/Multi-Tenancy) ← Can be deferred
|
|
||||||
↓
|
|
||||||
Phase 6 (Extended Services) → Phase 8 (Advanced Memory)
|
|
||||||
↓
|
|
||||||
Phase 9 (Extended Staff) → Phase 10 (UX) → Phase 11 (Production)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Critical Path**: Phases 1 → 2 → 3 → 4 must be sequential
|
|
||||||
**Can Be Deferred**: Phase 5 (Database) until you need persistence
|
|
||||||
**Parallel Opportunities**: Phase 6 and 7 can overlap; Phase 9 and 10 ongoing
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Overall Timeline Estimate
|
|
||||||
|
|
||||||
**Minimum Viable Household** (Phases 1-4): **15-20 weeks**
|
|
||||||
- Working Steward → Butler → Expert Agents with real LLM
|
|
||||||
- In-memory state (no persistence needed yet)
|
|
||||||
- Core household functional
|
|
||||||
|
|
||||||
**With Persistence** (Phases 1-5): **18-24 weeks**
|
|
||||||
- Add database and multi-tenancy
|
|
||||||
- Conversations survive restarts
|
|
||||||
- Multiple users supported
|
|
||||||
|
|
||||||
**Full-Featured System** (Phases 1-9): **35-45 weeks**
|
|
||||||
- All services integrated
|
|
||||||
- Advanced memory and context
|
|
||||||
- Extended household staff
|
|
||||||
|
|
||||||
**Production-Ready** (All phases): **40-50 weeks**
|
|
||||||
- Polished UX
|
|
||||||
- Hardened for homelab deployment
|
|
||||||
- Fully documented
|
|
||||||
|
|
||||||
*Note: Timeline assumes consistent part-time development effort*
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Success Metrics
|
|
||||||
|
|
||||||
### Technical
|
|
||||||
- System implements PHILOSOPHY.md patterns
|
|
||||||
- All household roles functional
|
|
||||||
- Multi-tenant isolation verified
|
|
||||||
- Real-time reasoning transparency working
|
|
||||||
- MCP integration complete
|
|
||||||
|
|
||||||
### User Experience
|
|
||||||
- Tatlock feels like interacting with a butler
|
|
||||||
- Wait times are transparent and acceptable
|
|
||||||
- Expert agents provide value in their domains
|
|
||||||
- System is reliable and trustworthy
|
|
||||||
|
|
||||||
### Architecture
|
|
||||||
- Clean separation between household roles
|
|
||||||
- Easy to add new agents/tools
|
|
||||||
- Model efficiency (base model stays loaded)
|
|
||||||
- Scales to household + friends usage
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Risk Management
|
|
||||||
|
|
||||||
### High Risk Items
|
|
||||||
1. **PydanticAI + Ollama integration complexity**
|
|
||||||
- Mitigation: Prototype early, iterate on connection layer
|
|
||||||
|
|
||||||
2. **Multi-agent coordination complexity**
|
|
||||||
- Mitigation: Start simple, add coordination gradually
|
|
||||||
|
|
||||||
3. **Model performance on homelab hardware**
|
|
||||||
- Mitigation: Model selection, quantization, optimization
|
|
||||||
|
|
||||||
4. **Prompt engineering for personality consistency**
|
|
||||||
- Mitigation: Extensive testing, user feedback, iteration
|
|
||||||
|
|
||||||
### Medium Risk Items
|
|
||||||
- MCP protocol adoption and tooling maturity
|
|
||||||
- Vector embedding quality for memory
|
|
||||||
- Home automation integration variability
|
|
||||||
- User authentication security
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Next Steps
|
|
||||||
|
|
||||||
1. **Immediate**: Commit model name fix (Tatlock)
|
|
||||||
2. **Week 1-2**: Begin Phase 1 (PostgreSQL + multi-tenancy design)
|
|
||||||
3. **Week 3**: Parallel prototype of Steward agent
|
|
||||||
4. **Ongoing**: Update this roadmap as we learn
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Document Status**: Active planning document
|
|
||||||
**Created**: 2025-12-06
|
|
||||||
**Last Updated**: 2025-12-06
|
|
||||||
@@ -0,0 +1,89 @@
|
|||||||
|
.PHONY: help setup run test test-unit test-integration test-contracts lint typecheck clean
|
||||||
|
|
||||||
|
VENV := .venv
|
||||||
|
PYTHON := $(VENV)/bin/python
|
||||||
|
PIP := $(VENV)/bin/pip
|
||||||
|
PYTEST := $(VENV)/bin/pytest
|
||||||
|
RUFF := $(VENV)/bin/ruff
|
||||||
|
MYPY := $(VENV)/bin/mypy
|
||||||
|
UVICORN := $(VENV)/bin/uvicorn
|
||||||
|
|
||||||
|
HOST := 0.0.0.0
|
||||||
|
PORT := 8777
|
||||||
|
|
||||||
|
help: ## Show this help
|
||||||
|
@grep -E '^[a-zA-Z_-]+:.*?## .*$$' $(MAKEFILE_LIST) | sort | awk 'BEGIN {FS = ":.*?## "}; {printf "\033[36m%-20s\033[0m %s\n", $$1, $$2}'
|
||||||
|
|
||||||
|
setup: ## Create venv and install all dependencies
|
||||||
|
python3 -m venv $(VENV)
|
||||||
|
$(PIP) install --upgrade pip
|
||||||
|
$(PIP) install -e ".[dev]"
|
||||||
|
# Exit 0 from pip install is not evidence the environment works (D-24) - the
|
||||||
|
# 2026-08-09 core-api incident was exactly this: a venv that "installed fine"
|
||||||
|
# but was missing a declared dependency, surfacing as 11 collection errors
|
||||||
|
# that read like broken imports rather than an environment problem. Collection
|
||||||
|
# is the right cheap check here for that same reason: it imports every test
|
||||||
|
# module (and everything they import) without running the suite, so a missing
|
||||||
|
# or mismatched dependency fails setup itself instead of showing up later as a
|
||||||
|
# mysterious test failure. Scoped like `make test` (excludes e2e/integration/
|
||||||
|
# contracts, which need external services) and --no-cov since coverage
|
||||||
|
# instrumentation is irrelevant to "does this collect".
|
||||||
|
$(PYTEST) --collect-only -q --ignore=tests/e2e --ignore=tests/integration --ignore=tests/contracts --no-cov
|
||||||
|
|
||||||
|
run: ## Start the development server on port 8777
|
||||||
|
@mkdir -p build/logs
|
||||||
|
@if lsof -Pi :$(PORT) -sTCP:LISTEN -t >/dev/null 2>&1; then \
|
||||||
|
echo "Error: Port $(PORT) is already in use"; \
|
||||||
|
echo "Run: lsof -i :$(PORT) to see what's using it"; \
|
||||||
|
exit 1; \
|
||||||
|
fi
|
||||||
|
$(UVICORN) src.main:app --reload --host $(HOST) --port $(PORT) 2>&1 | tee build/logs/server.log
|
||||||
|
|
||||||
|
test: ## Run unit tests (no external services needed)
|
||||||
|
$(PYTEST) --ignore=tests/e2e --ignore=tests/integration --ignore=tests/contracts
|
||||||
|
|
||||||
|
test-unit: test ## Alias for test
|
||||||
|
|
||||||
|
test-integration: ## Run integration tests (needs Claude/Ollama)
|
||||||
|
$(PYTEST) tests/agents/test_tatlock_agent.py -v
|
||||||
|
|
||||||
|
test-contracts: ## Wire-level contract tests against live service boundaries
|
||||||
|
$(PYTEST) tests/contracts -v --no-cov
|
||||||
|
|
||||||
|
lint: ## Run ruff linter and formatter check
|
||||||
|
$(RUFF) check src tests
|
||||||
|
$(RUFF) format --check src tests
|
||||||
|
|
||||||
|
typecheck: ## Run mypy type checking
|
||||||
|
$(MYPY) src
|
||||||
|
|
||||||
|
clean: ## Remove build artifacts, caches, and coverage reports
|
||||||
|
rm -rf .cache build
|
||||||
|
find . -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true
|
||||||
|
|
||||||
|
# git hands a hook a non-login shell, which never sees ~/.local/bin — where
|
||||||
|
# gitleaks lands. Without this the scan reports "not installed" on every push,
|
||||||
|
# which is a check that fails open (D-24).
|
||||||
|
export PATH := $(HOME)/.local/bin:/usr/local/bin:$(PATH)
|
||||||
|
|
||||||
|
.PHONY: secrets
|
||||||
|
secrets: ## Scan the commits about to be pushed for credentials
|
||||||
|
@ci/secrets.sh
|
||||||
|
|
||||||
|
# The call surface is identical in every repo; what it runs is not.
|
||||||
|
#
|
||||||
|
# `secrets` runs first, deliberately: it is the only failure here that cannot be
|
||||||
|
# undone by fixing it afterwards. A failed lint costs another commit; a pushed
|
||||||
|
# credential is cached and indexed whether or not it is later deleted.
|
||||||
|
#
|
||||||
|
# Some of these fail today, and are left wired anyway. The state was measured
|
||||||
|
# once and written down in T-56 rather than being worked around here — a gate
|
||||||
|
# quietly narrowed to what already passes is a gate that reports success for
|
||||||
|
# doing nothing, which is the failure this workspace keeps rediscovering.
|
||||||
|
.PHONY: pre-push
|
||||||
|
pre-push: secrets lint ## Everything the pre-push hook runs
|
||||||
|
@echo " -- not gated here yet: typecheck (T-1), test (T-56)"
|
||||||
|
@echo " typecheck reports 95 errors in 31 files and has never passed, so"
|
||||||
|
@echo " gating on it blocked every push to this repo — including the commit"
|
||||||
|
@echo " that added the gate. Run 'make typecheck' before pushing anything"
|
||||||
|
@echo " that touches types; T-1 is the pass that earns this line's removal."
|
||||||
@@ -1,535 +0,0 @@
|
|||||||
# Phase 2 Completion Summary: The Steward
|
|
||||||
|
|
||||||
**Status**: ✅ COMPLETE
|
|
||||||
**Completed**: 2025-12-07
|
|
||||||
**Duration**: 1 day (accelerated from 7-week plan)
|
|
||||||
**Test Coverage**: 223 passing tests (99.5% pass rate)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Executive Summary
|
|
||||||
|
|
||||||
Phase 2 successfully implements **The Steward** - a first-tier LLM agent that creates a two-tier architecture for intelligent request routing. The Steward analyzes incoming requests, identifies relevant household capabilities, and provides scoped tool recommendations to Tatlock (the Butler).
|
|
||||||
|
|
||||||
This architecture prevents cognitive overload by ensuring Tatlock only sees tools relevant to each specific request, while maintaining full conversation context awareness and providing complete observability through benchmarking and logging.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Delivered Features
|
|
||||||
|
|
||||||
### 1. The Steward Agent ✅
|
|
||||||
**Location**: `src/agents/steward/`
|
|
||||||
|
|
||||||
- **Request Analysis**: Analyzes user requests with full conversation history
|
|
||||||
- **Capability Recommendation**: Recommends relevant household tools/capabilities
|
|
||||||
- **Context Awareness**: Identifies references to previous conversation turns
|
|
||||||
- **Complexity Assessment**: Estimates request complexity (simple/moderate/complex)
|
|
||||||
- **Missing Capability Detection**: Explicitly states when needed tools are unavailable
|
|
||||||
- **VRAM Efficiency**: Uses same Ollama model as Tatlock (mistral-nemo:latest)
|
|
||||||
|
|
||||||
**Key Files**:
|
|
||||||
- `agent.py`: Steward PydanticAI agent implementation
|
|
||||||
- `schemas.py`: `StewardRecommendation` and `ConversationContext` structures
|
|
||||||
- `service.py`: Service layer with logging and benchmarking
|
|
||||||
|
|
||||||
### 2. Household Registry ✅
|
|
||||||
**Location**: `src/core/household_registry.py`
|
|
||||||
|
|
||||||
- **Centralized Capability Management**: Single source of truth for household tools
|
|
||||||
- **Executive Summaries**: High-level capability descriptions for Steward/Butler coordination
|
|
||||||
- **PydanticAI Toolsets**: Native toolset composition and scoping
|
|
||||||
- **Domain Organization**: Tools organized by household member (e.g., `tatlock_core`)
|
|
||||||
- **Dynamic Tool Scoping**: Creates combined toolsets based on recommendations
|
|
||||||
|
|
||||||
**Architecture**:
|
|
||||||
```
|
|
||||||
HouseholdRegistry
|
|
||||||
├─ HouseholdMember (tatlock_core)
|
|
||||||
│ ├─ HouseholdCapability (summary)
|
|
||||||
│ └─ FunctionToolset (calculator, datetime, search)
|
|
||||||
├─ Future: HouseholdMember (librarian)
|
|
||||||
└─ Future: HouseholdMember (developer)
|
|
||||||
```
|
|
||||||
|
|
||||||
### 3. Request Preprocessing Pipeline ✅
|
|
||||||
**Location**: `src/core/preprocessing.py`
|
|
||||||
|
|
||||||
**4-Phase Flow**:
|
|
||||||
1. **Steward Analysis**: Analyzes request with full conversation history
|
|
||||||
2. **Tool Scoping**: Creates combined toolset from recommendations
|
|
||||||
3. **Note Formatting**: Prepares Steward note for Butler (invisible to user)
|
|
||||||
4. **Enrichment**: Returns `EnrichedRequest` with all context
|
|
||||||
|
|
||||||
**Integration**: Fully integrated with Responses API via `create_response_with_steward()`
|
|
||||||
|
|
||||||
### 4. Tool Usage Tracking ✅
|
|
||||||
**Location**: `src/core/tool_tracking.py`
|
|
||||||
|
|
||||||
**Capabilities**:
|
|
||||||
- Tracks recommended vs. actual tool usage
|
|
||||||
- Logs unexpected tool calls (not recommended but used)
|
|
||||||
- Logs unused recommendations (recommended but not used)
|
|
||||||
- Records timing data for each tool call
|
|
||||||
- Stores benchmarks to Redis for analysis
|
|
||||||
|
|
||||||
**Metrics Supported**:
|
|
||||||
- Precision: Recommended and used / All recommendations
|
|
||||||
- Recall: Recommended and used / All tool calls
|
|
||||||
- F1 Score: Harmonic mean of precision and recall
|
|
||||||
|
|
||||||
### 5. Streaming Transparency ✅
|
|
||||||
**Location**: `src/responses/streaming.py`
|
|
||||||
|
|
||||||
**Features**:
|
|
||||||
- Streams Steward's analysis first (reasoning summary deltas)
|
|
||||||
- Streams Tatlock's response second (output text deltas)
|
|
||||||
- Full SSE support with proper event types
|
|
||||||
- Conversation context visible in stream
|
|
||||||
- Missing capabilities warnings included
|
|
||||||
|
|
||||||
**Event Sequence**:
|
|
||||||
```
|
|
||||||
1. response.reasoning_summary_text.delta (Steward analysis)
|
|
||||||
2. response.reasoning_summary_text.done
|
|
||||||
3. response.output_text.delta (Tatlock response)
|
|
||||||
4. response.output_text.done
|
|
||||||
5. response.done (final response)
|
|
||||||
```
|
|
||||||
|
|
||||||
### 6. Structured Logging ✅
|
|
||||||
**Location**: `src/core/logging_config.py`
|
|
||||||
|
|
||||||
**Features**:
|
|
||||||
- JSON-formatted structured logging via `structlog`
|
|
||||||
- Operation timing via context managers (`log_operation`)
|
|
||||||
- Metadata enrichment for debugging
|
|
||||||
- Integrated with benchmark recording
|
|
||||||
- Machine-parseable output for analysis
|
|
||||||
|
|
||||||
### 7. Redis Benchmark Storage ✅
|
|
||||||
**Location**: `src/core/benchmarks.py`
|
|
||||||
|
|
||||||
**Features**:
|
|
||||||
- Cross-session performance metrics storage
|
|
||||||
- Time-series data with 30-day automatic expiry
|
|
||||||
- Operations tracked: `steward_analysis`, `tool_call`
|
|
||||||
- Queryable by operation type, time range, metadata
|
|
||||||
- Supports accuracy analysis (recommended vs. used)
|
|
||||||
|
|
||||||
**Benchmark Schema**:
|
|
||||||
- Timestamp, operation, duration, success/failure
|
|
||||||
- Steward-specific: recommendation_count, complexity
|
|
||||||
- Tool-specific: tool_name, was_recommended, was_actually_used
|
|
||||||
- Context: conversation_id, metadata dict
|
|
||||||
|
|
||||||
### 8. Benchmark Analysis Tools ✅
|
|
||||||
**Location**: `scripts/benchmark_analysis.py`
|
|
||||||
|
|
||||||
**CLI Features**:
|
|
||||||
```bash
|
|
||||||
# Steward performance over last 24 hours
|
|
||||||
python scripts/benchmark_analysis.py --operation steward_analysis --hours 24
|
|
||||||
|
|
||||||
# Tool recommendation accuracy over last 7 days
|
|
||||||
python scripts/benchmark_analysis.py --tool-accuracy --days 7
|
|
||||||
|
|
||||||
# Summary of all operations
|
|
||||||
python scripts/benchmark_analysis.py --summary --hours 1
|
|
||||||
```
|
|
||||||
|
|
||||||
**Metrics Provided**:
|
|
||||||
- Average Steward latency (target: < 2s)
|
|
||||||
- Success rate percentage
|
|
||||||
- Recommendation count distribution
|
|
||||||
- Complexity distribution
|
|
||||||
- Tool-specific accuracy (precision/recall/F1)
|
|
||||||
- Per-tool usage patterns
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Architecture
|
|
||||||
|
|
||||||
### Request Flow
|
|
||||||
|
|
||||||
```
|
|
||||||
User Request
|
|
||||||
↓
|
|
||||||
Responses API (FastAPI)
|
|
||||||
↓
|
|
||||||
┌─────────────────────────────────────────────┐
|
|
||||||
│ Preprocessing Pipeline │
|
|
||||||
│ ├─ Steward Agent │
|
|
||||||
│ │ ├─ Receives: Full conversation history │
|
|
||||||
│ │ ├─ Analyzes: Context + requirements │
|
|
||||||
│ │ ├─ Queries: Household registry │
|
|
||||||
│ │ └─ Returns: StewardRecommendation │
|
|
||||||
│ │ │
|
|
||||||
│ ├─ Create Scoped Toolset │
|
|
||||||
│ │ └─ CombinedToolset from capabilities │
|
|
||||||
│ │ │
|
|
||||||
│ └─ Format Steward Note │
|
|
||||||
│ └─ Context summary for Butler │
|
|
||||||
└─────────────────────────────────────────────┘
|
|
||||||
↓
|
|
||||||
Tatlock Agent (Butler)
|
|
||||||
├─ Receives: Enriched request + note
|
|
||||||
├─ Tools: ONLY scoped recommendations
|
|
||||||
├─ Tracking: Tool usage monitored
|
|
||||||
└─ Context: Full conversation history
|
|
||||||
↓
|
|
||||||
Response to User
|
|
||||||
├─ Steward's reasoning (streamed first)
|
|
||||||
└─ Tatlock's response (streamed second)
|
|
||||||
|
|
||||||
Background:
|
|
||||||
└─ Redis: Benchmarks + metrics
|
|
||||||
```
|
|
||||||
|
|
||||||
### Two-Tier Abstraction
|
|
||||||
|
|
||||||
**Tier 1: Executive Summaries (Steward/Butler coordination)**
|
|
||||||
```python
|
|
||||||
HouseholdCapability(
|
|
||||||
name="tatlock_core",
|
|
||||||
role="Butler's Core Tools",
|
|
||||||
category="core",
|
|
||||||
description="Mathematical calculation, date/time operations, web search",
|
|
||||||
domains=["computation", "information", "datetime"],
|
|
||||||
cost="low",
|
|
||||||
requires_network=True
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Tier 2: Implementation Details (Tool execution)**
|
|
||||||
```python
|
|
||||||
FunctionToolset containing:
|
|
||||||
- calculate(expression: str) -> str
|
|
||||||
- get_current_datetime(format_str: str) -> str
|
|
||||||
- calculate_time_offset(offset: str) -> str
|
|
||||||
- time_difference(date1: str, date2: str) -> str
|
|
||||||
- search_web(query: str, num_results: int) -> str
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Test Coverage
|
|
||||||
|
|
||||||
### Test Statistics
|
|
||||||
- **Total Tests**: 223 (219 passing, 1 pre-existing failure unrelated to Phase 2)
|
|
||||||
- **Pass Rate**: 99.5%
|
|
||||||
- **Coverage**: 77.6% overall
|
|
||||||
|
|
||||||
### Test Categories
|
|
||||||
|
|
||||||
#### Unit Tests ✅
|
|
||||||
- **Household Registry** (12 tests): Registration, retrieval, toolset composition
|
|
||||||
- **Steward Schemas** (11 tests): Data structures, formatting
|
|
||||||
- **Steward Service** (9 tests): Request analysis, context detection, capabilities
|
|
||||||
- **Preprocessing** (6 tests via integration): Request enrichment, tool scoping
|
|
||||||
|
|
||||||
#### Integration Tests ✅
|
|
||||||
- **Steward → Tatlock Flow** (6 tests):
|
|
||||||
- Simple math request
|
|
||||||
- Conversation history propagation
|
|
||||||
- No capabilities needed (conversational)
|
|
||||||
- Tool tracker integration
|
|
||||||
- Missing capabilities warning
|
|
||||||
- Conversation ID propagation
|
|
||||||
|
|
||||||
- **Streaming Integration** (4 tests):
|
|
||||||
- Basic streaming with Steward
|
|
||||||
- Conversation history in streaming
|
|
||||||
- Reasoning contains Steward analysis
|
|
||||||
- Missing capabilities in stream
|
|
||||||
|
|
||||||
### Key Test Files
|
|
||||||
- `tests/agents/steward/test_steward_schemas.py`
|
|
||||||
- `tests/agents/steward/test_steward_service.py`
|
|
||||||
- `tests/integration/test_steward_tatlock_integration.py`
|
|
||||||
- `tests/integration/test_steward_streaming.py`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Technical Achievements
|
|
||||||
|
|
||||||
### 1. PydanticAI Native Patterns ✅
|
|
||||||
- `FunctionToolset` for tool grouping
|
|
||||||
- `CombinedToolset` for dynamic composition
|
|
||||||
- Decorator-based tool registration (`@agent.tool`)
|
|
||||||
- Structured outputs via Pydantic models (`StewardRecommendation`)
|
|
||||||
- Dependency injection for tracking (`RunContext[ToolCallTracker]`)
|
|
||||||
|
|
||||||
### 2. Tool Scoping Enforcement ✅
|
|
||||||
- Compile-time scoping via toolset creation
|
|
||||||
- Tools not even visible to LLM if not recommended
|
|
||||||
- Fresh agent instances with scoped tools only
|
|
||||||
- No runtime permission checks needed
|
|
||||||
|
|
||||||
### 3. Conversation Context Awareness ✅
|
|
||||||
- Steward sees FULL conversation history
|
|
||||||
- Identifies references to previous turns
|
|
||||||
- Provides contextual notes to Butler
|
|
||||||
- Example: "User mentioned Python debugging in turn 3"
|
|
||||||
|
|
||||||
### 4. Plain Text Approach ✅
|
|
||||||
- Steward returns natural language analysis
|
|
||||||
- Service layer parses for structured data
|
|
||||||
- Keyword extraction for capabilities
|
|
||||||
- Pattern matching for complexity and context
|
|
||||||
|
|
||||||
### 5. Observability ✅
|
|
||||||
- Structured logging for all operations
|
|
||||||
- Benchmark recording to Redis
|
|
||||||
- Tool usage tracking (recommended vs. actual)
|
|
||||||
- Cross-session performance analysis
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Performance Characteristics
|
|
||||||
|
|
||||||
### Latency (Estimated)
|
|
||||||
- **Steward Analysis**: ~1-2 seconds (single LLM call)
|
|
||||||
- **Tatlock Execution**: ~2-5 seconds (depends on tool usage)
|
|
||||||
- **Total Added Overhead**: ~1-2 seconds vs. direct Tatlock call
|
|
||||||
- **Streaming Transparency**: Steward reasoning visible immediately
|
|
||||||
|
|
||||||
### Resource Usage
|
|
||||||
- **VRAM**: Same model for both agents (mistral-nemo:latest)
|
|
||||||
- **Model Loading**: No additional model loads (efficient!)
|
|
||||||
- **Redis**: Minimal (benchmarks with 30-day expiry)
|
|
||||||
- **Network**: Only when web search tools used
|
|
||||||
|
|
||||||
### Accuracy Targets
|
|
||||||
- **Recommendation Precision**: > 90% (tools recommended and actually used)
|
|
||||||
- **Recommendation Recall**: > 90% (tools used were recommended)
|
|
||||||
- **False Positives**: < 10% (recommended but not used)
|
|
||||||
- **False Negatives**: < 10% (used but not recommended)
|
|
||||||
|
|
||||||
*Note: Actual metrics available via `scripts/benchmark_analysis.py` after production usage*
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Files Created
|
|
||||||
|
|
||||||
### Core Implementation
|
|
||||||
1. `src/core/household_registry.py` - Capability management
|
|
||||||
2. `src/core/preprocessing.py` - Request preprocessing pipeline
|
|
||||||
3. `src/core/tool_tracking.py` - Tool usage tracking
|
|
||||||
4. `src/core/logging_config.py` - Structured logging (M1)
|
|
||||||
5. `src/core/benchmarks.py` - Redis benchmark storage (M1)
|
|
||||||
|
|
||||||
### Steward Agent
|
|
||||||
6. `src/agents/steward/agent.py` - Steward PydanticAI agent
|
|
||||||
7. `src/agents/steward/schemas.py` - Data structures
|
|
||||||
8. `src/agents/steward/service.py` - Service layer
|
|
||||||
|
|
||||||
### Tatlock Core Organization
|
|
||||||
9. `src/agents/tatlock_core/tools.py` - Tool implementations (reorganized)
|
|
||||||
10. `src/agents/tatlock_core/toolset.py` - PydanticAI toolset
|
|
||||||
11. `src/agents/tatlock_core/capability.py` - Registry integration
|
|
||||||
|
|
||||||
### Tests
|
|
||||||
12. `tests/agents/steward/test_steward_schemas.py` - Schema tests
|
|
||||||
13. `tests/agents/steward/test_steward_service.py` - Service tests
|
|
||||||
14. `tests/integration/test_steward_tatlock_integration.py` - Full flow tests
|
|
||||||
15. `tests/integration/test_steward_streaming.py` - Streaming tests
|
|
||||||
|
|
||||||
### Tools & Documentation
|
|
||||||
16. `scripts/benchmark_analysis.py` - Performance analysis CLI
|
|
||||||
17. `PHASE2_PLAN.md` - Detailed implementation plan
|
|
||||||
18. `PHASE2_COMPLETE.md` - This completion summary
|
|
||||||
|
|
||||||
### Modified Files
|
|
||||||
- `src/agents/tatlock.py` - Added `run_with_scoped_tools()` method
|
|
||||||
- `src/responses/service.py` - Added `create_response_with_steward()`
|
|
||||||
- `src/responses/router.py` - Steward routing logic
|
|
||||||
- `src/responses/streaming.py` - Added `stream_response_with_steward()`
|
|
||||||
- `CHANGELOG.md` - Phase 2 documentation
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Success Metrics
|
|
||||||
|
|
||||||
### Technical ✅
|
|
||||||
- ✅ Household registry operational with executive summaries
|
|
||||||
- ✅ Steward produces structured recommendations
|
|
||||||
- ✅ Steward analyzes full conversation context
|
|
||||||
- ✅ Tool scoping enforced (Tatlock can't use non-recommended tools)
|
|
||||||
- ✅ Model efficiency preserved (no reload delays)
|
|
||||||
- ✅ Performance benchmarks recorded to Redis
|
|
||||||
- ✅ Tool usage tracking (recommended vs. actual)
|
|
||||||
- ✅ Streaming transparency implemented
|
|
||||||
|
|
||||||
### Observability ✅
|
|
||||||
- ✅ Structured logging (JSON format)
|
|
||||||
- ✅ Benchmark analysis tools available
|
|
||||||
- ✅ Tool recommendation accuracy measurable
|
|
||||||
- ✅ Cross-session performance trends visible
|
|
||||||
|
|
||||||
### Architectural ✅
|
|
||||||
- ✅ PydanticAI patterns followed (Toolsets, decorators, structured outputs)
|
|
||||||
- ✅ Clean separation: registry vs. agents vs. tools
|
|
||||||
- ✅ Two-tier abstraction working (summaries vs. details)
|
|
||||||
- ✅ Future-proof for expert agents (Phase 4)
|
|
||||||
|
|
||||||
### Testing ✅
|
|
||||||
- ✅ 223 tests passing (99.5% pass rate)
|
|
||||||
- ✅ Integration tests for full flow
|
|
||||||
- ✅ Streaming integration tests
|
|
||||||
- ✅ 77.6% test coverage maintained
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Usage Examples
|
|
||||||
|
|
||||||
### Non-Streaming Request
|
|
||||||
```python
|
|
||||||
from src.responses.service import create_response_with_steward
|
|
||||||
from src.responses.schemas import ResponseRequest
|
|
||||||
|
|
||||||
request = ResponseRequest(
|
|
||||||
model="tatlock",
|
|
||||||
input=[
|
|
||||||
{"role": "user", "content": "What's sqrt(144)?"}
|
|
||||||
],
|
|
||||||
metadata={"conversation_id": "conv_123"}
|
|
||||||
)
|
|
||||||
|
|
||||||
response = await create_response_with_steward(request)
|
|
||||||
|
|
||||||
# Response includes:
|
|
||||||
# 1. Steward's analysis (reasoning output)
|
|
||||||
# 2. Tatlock's answer (message output)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Streaming Request
|
|
||||||
```python
|
|
||||||
from src.responses.streaming import StreamingCoordinator
|
|
||||||
|
|
||||||
coordinator = StreamingCoordinator()
|
|
||||||
|
|
||||||
async for event in coordinator.stream_response_with_steward(request):
|
|
||||||
if event.event == "response.reasoning_summary_text.delta":
|
|
||||||
print(f"Steward: {event.delta}", end="")
|
|
||||||
elif event.event == "response.output_text.delta":
|
|
||||||
print(f"Tatlock: {event.delta}", end="")
|
|
||||||
elif event.event == "response.done":
|
|
||||||
print(f"\nFinal response: {event.response.id}")
|
|
||||||
```
|
|
||||||
|
|
||||||
### Benchmark Analysis
|
|
||||||
```bash
|
|
||||||
# View Steward performance
|
|
||||||
python scripts/benchmark_analysis.py --operation steward_analysis --hours 24
|
|
||||||
|
|
||||||
# Analyze tool accuracy
|
|
||||||
python scripts/benchmark_analysis.py --tool-accuracy --days 7
|
|
||||||
|
|
||||||
# Get summary
|
|
||||||
python scripts/benchmark_analysis.py --summary --hours 1
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Future-Proofing for Phase 4
|
|
||||||
|
|
||||||
### Expert Agent Pattern (Ready to Use)
|
|
||||||
|
|
||||||
When adding The Librarian, The Developer, or other expert agents:
|
|
||||||
|
|
||||||
```
|
|
||||||
src/agents/librarian/
|
|
||||||
├── agent.py # Librarian PydanticAI agent
|
|
||||||
├── tools.py # Research, wiki, knowledge tools
|
|
||||||
├── toolset.py # PydanticAI toolset
|
|
||||||
└── capability.py # Registry integration
|
|
||||||
```
|
|
||||||
|
|
||||||
**Registration**:
|
|
||||||
```python
|
|
||||||
from src.core.household_registry import get_household_registry
|
|
||||||
|
|
||||||
registry = get_household_registry()
|
|
||||||
registry.register(
|
|
||||||
name="librarian",
|
|
||||||
capability=LIBRARIAN_CAPABILITY,
|
|
||||||
toolset=librarian_toolset,
|
|
||||||
agent=librarian_agent # For delegation
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Delegation from Tatlock** (Phase 4):
|
|
||||||
```python
|
|
||||||
@tatlock_agent.tool
|
|
||||||
async def consult_librarian(
|
|
||||||
ctx: RunContext[None],
|
|
||||||
research_query: str
|
|
||||||
) -> str:
|
|
||||||
"""Consult the Librarian for research assistance."""
|
|
||||||
return await librarian_agent.run(research_query, usage=ctx.usage)
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Lessons Learned
|
|
||||||
|
|
||||||
### What Went Well
|
|
||||||
1. **PydanticAI Integration**: Native toolset patterns work beautifully
|
|
||||||
2. **Two-Tier Architecture**: Clean separation between coordination and execution
|
|
||||||
3. **Plain Text Approach**: More flexible than structured output for Steward
|
|
||||||
4. **Test Coverage**: Comprehensive integration tests caught edge cases early
|
|
||||||
5. **Streaming**: SSE events provide excellent real-time transparency
|
|
||||||
|
|
||||||
### Challenges Overcome
|
|
||||||
1. **Schema vs. Agent OutputItems**: Fixed `_calculate_usage` to handle both types
|
|
||||||
2. **Registry Initialization**: Added fixtures to ensure registry available in tests
|
|
||||||
3. **Plain Text Parsing**: Keyword extraction works well but needs careful test mocking
|
|
||||||
4. **Complexity Substring Matching**: "Complexity:" contains "complex" - fixed test mocks
|
|
||||||
|
|
||||||
### Optimizations
|
|
||||||
1. **Single Model**: Using same Ollama model for both agents saves VRAM
|
|
||||||
2. **Sequential Execution**: No parallel LLM calls needed (Steward → Tatlock)
|
|
||||||
3. **Tool Scoping**: Fresh agent instances more reliable than runtime filtering
|
|
||||||
4. **Benchmark Expiry**: 30-day TTL prevents Redis bloat
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Next Steps
|
|
||||||
|
|
||||||
### Immediate
|
|
||||||
- Monitor Steward accuracy in production
|
|
||||||
- Collect real-world benchmarks
|
|
||||||
- Iterate on Steward prompt based on metrics
|
|
||||||
|
|
||||||
### Phase 3 (Optional)
|
|
||||||
- Web search delegation to The Librarian
|
|
||||||
- Enhanced research capabilities
|
|
||||||
- Multi-source information synthesis
|
|
||||||
|
|
||||||
### Phase 4
|
|
||||||
- Expert agent delegation (Librarian, Developer, etc.)
|
|
||||||
- Dynamic agent selection based on request
|
|
||||||
- Cross-agent collaboration patterns
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Conclusion
|
|
||||||
|
|
||||||
Phase 2 successfully delivers a production-ready two-tier architecture with The Steward managing intelligent request routing and tool scoping. The implementation is:
|
|
||||||
|
|
||||||
- ✅ **Complete**: All planned features delivered
|
|
||||||
- ✅ **Tested**: 223 tests with 99.5% pass rate
|
|
||||||
- ✅ **Observable**: Full logging and benchmarking
|
|
||||||
- ✅ **Efficient**: Single model, minimal overhead
|
|
||||||
- ✅ **Extensible**: Ready for expert agents in Phase 4
|
|
||||||
|
|
||||||
The Steward provides intelligent capability coordination while maintaining conversation context awareness, creating a foundation for scalable multi-agent collaboration in future phases.
|
|
||||||
|
|
||||||
**Phase 2 Status**: ✅ **COMPLETE**
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Document Version**: 1.0
|
|
||||||
**Created**: 2025-12-07
|
|
||||||
**Author**: Development Team
|
|
||||||
**Reference**: [PHASE2_PLAN.md](PHASE2_PLAN.md)
|
|
||||||
-865
@@ -1,865 +0,0 @@
|
|||||||
# Phase 2 Implementation Plan: The Steward
|
|
||||||
|
|
||||||
**Status**: Active Planning
|
|
||||||
**Created**: 2025-12-07
|
|
||||||
**Estimated Duration**: 4-5 weeks
|
|
||||||
**Goal**: Implement first-tier request analysis and household capability coordination
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Executive Summary
|
|
||||||
|
|
||||||
Phase 2 introduces **The Steward** - a first-tier LLM agent that analyzes incoming requests, identifies relevant household capabilities, and provides focused recommendations to Tatlock (the Butler). This creates a two-tier architecture that prevents cognitive overload and enables efficient tool/agent coordination.
|
|
||||||
|
|
||||||
### Key Deliverables
|
|
||||||
|
|
||||||
1. **Household Registry**: Centralized capability catalog with PydanticAI Toolsets
|
|
||||||
2. **Steward Agent**: Request analyzer with conversation context awareness
|
|
||||||
3. **Tool Scoping**: Dynamic toolset creation based on recommendations
|
|
||||||
4. **Observability**: Performance benchmarking and tool usage tracking via Redis
|
|
||||||
5. **Integration**: Full Steward → Tatlock request flow
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Core Architectural Principles
|
|
||||||
|
|
||||||
### 1. Household-Based Organization
|
|
||||||
- Each expert agent owns their tools in a domain directory
|
|
||||||
- Tools organized as functional clusters around capabilities
|
|
||||||
- Example: `src/agents/tatlock_core/` contains calculator, datetime, web search
|
|
||||||
|
|
||||||
### 2. Two-Tier Capability Abstraction
|
|
||||||
- **Executive Summary**: High-level capabilities for Steward/Butler coordination
|
|
||||||
- **Implementation Details**: Full tool specifications for household members
|
|
||||||
- Steward sees summaries, household members see full details
|
|
||||||
|
|
||||||
### 3. PydanticAI Native Patterns
|
|
||||||
- Use `FunctionToolset` and `CombinedToolset` for composition
|
|
||||||
- Decorator-based tool registration (`@agent.tool`)
|
|
||||||
- Structured outputs via Pydantic models
|
|
||||||
- Agent delegation pattern for expert agents (Phase 4)
|
|
||||||
|
|
||||||
### 4. Separate Registries
|
|
||||||
- **Household Registry**: Tools + capabilities (new in Phase 2)
|
|
||||||
- **Model Registry**: Agents/models (existing from Phase 1)
|
|
||||||
- Clean separation of concerns
|
|
||||||
|
|
||||||
### 5. Start Minimal
|
|
||||||
- Only 3 core Tatlock tools initially: calculator, datetime, web search
|
|
||||||
- No new tools until expert agents exist (Phase 4)
|
|
||||||
- Prove the pattern before expanding
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Implementation Milestones
|
|
||||||
|
|
||||||
|
|
||||||
### Milestone 1: Household Registry + Logging Infrastructure (Week 1-2)
|
|
||||||
|
|
||||||
#### Goal
|
|
||||||
Create a registry system that aggregates household capabilities using PydanticAI Toolsets and establish observability infrastructure.
|
|
||||||
|
|
||||||
#### Tasks
|
|
||||||
|
|
||||||
**1.1 Create Household Registry Module**
|
|
||||||
|
|
||||||
Location: `src/core/household_registry.py`
|
|
||||||
|
|
||||||
```python
|
|
||||||
from pydantic import BaseModel
|
|
||||||
from pydantic_ai import FunctionToolset, CombinedToolset
|
|
||||||
|
|
||||||
class HouseholdCapability(BaseModel):
|
|
||||||
"""Executive summary of a household member's capabilities."""
|
|
||||||
name: str # "tatlock_core", "librarian", "developer"
|
|
||||||
role: str # "Butler's Core Tools", "The Librarian"
|
|
||||||
category: str # "core", "research", "technical"
|
|
||||||
description: str # One-sentence description
|
|
||||||
domains: list[str] # ["computation", "information", "datetime"]
|
|
||||||
cost: str # "low", "medium", "high"
|
|
||||||
requires_network: bool
|
|
||||||
|
|
||||||
class HouseholdMember(BaseModel):
|
|
||||||
"""Full specification of a household member."""
|
|
||||||
capability: HouseholdCapability
|
|
||||||
toolset: FunctionToolset
|
|
||||||
agent: Agent | None = None # For expert agents in Phase 4
|
|
||||||
|
|
||||||
class HouseholdRegistry:
|
|
||||||
"""Registry of household capabilities and implementations."""
|
|
||||||
|
|
||||||
def __init__(self):
|
|
||||||
self._members: dict[str, HouseholdMember] = {}
|
|
||||||
|
|
||||||
def register(
|
|
||||||
self,
|
|
||||||
name: str,
|
|
||||||
capability: HouseholdCapability,
|
|
||||||
toolset: FunctionToolset,
|
|
||||||
agent: Agent | None = None
|
|
||||||
):
|
|
||||||
"""Register a household member."""
|
|
||||||
self._members[name] = HouseholdMember(
|
|
||||||
capability=capability,
|
|
||||||
toolset=toolset,
|
|
||||||
agent=agent
|
|
||||||
)
|
|
||||||
|
|
||||||
def get_all_capabilities(self) -> list[HouseholdCapability]:
|
|
||||||
"""Get executive summaries for Steward/Butler."""
|
|
||||||
return [m.capability for m in self._members.values()]
|
|
||||||
|
|
||||||
def get_scoped_toolset(self, names: list[str]) -> CombinedToolset:
|
|
||||||
"""Create combined toolset from recommended capabilities."""
|
|
||||||
toolsets = [self._members[name].toolset for name in names]
|
|
||||||
return CombinedToolset(toolsets)
|
|
||||||
|
|
||||||
# Global registry instance
|
|
||||||
household_registry = HouseholdRegistry()
|
|
||||||
```
|
|
||||||
|
|
||||||
|
|
||||||
**1.2 Reorganize Tatlock Core Tools**
|
|
||||||
|
|
||||||
Create domain-based organization:
|
|
||||||
|
|
||||||
```
|
|
||||||
src/agents/tatlock_core/
|
|
||||||
├── __init__.py
|
|
||||||
├── tools.py # Tool implementations (moved from src/agents/tools.py)
|
|
||||||
├── toolset.py # PydanticAI toolset registration
|
|
||||||
└── capability.py # Executive summary for registry
|
|
||||||
```
|
|
||||||
|
|
||||||
**1.3 Create Logging Infrastructure**
|
|
||||||
|
|
||||||
Location: `src/core/logging_config.py`
|
|
||||||
|
|
||||||
- Structured logging with `structlog`
|
|
||||||
- JSON format for machine parsing
|
|
||||||
- Operation timing and metadata tracking
|
|
||||||
- Context manager for automatic timing
|
|
||||||
|
|
||||||
**1.4 Create Redis Benchmark Storage**
|
|
||||||
|
|
||||||
Location: `src/core/benchmarks.py`
|
|
||||||
|
|
||||||
Features:
|
|
||||||
- Performance benchmark recording (Steward analysis, tool calls)
|
|
||||||
- Cross-session persistence via Redis
|
|
||||||
- Time-series storage with automatic expiry (30 days)
|
|
||||||
- Queryable metrics for analysis
|
|
||||||
|
|
||||||
Benchmark schema:
|
|
||||||
```python
|
|
||||||
class PerformanceBenchmark(BaseModel):
|
|
||||||
timestamp: datetime
|
|
||||||
operation: str # "steward_analysis", "tool_call"
|
|
||||||
duration_seconds: float
|
|
||||||
success: bool
|
|
||||||
|
|
||||||
# Steward-specific
|
|
||||||
recommendation_count: Optional[int]
|
|
||||||
confidence: Optional[float]
|
|
||||||
|
|
||||||
# Tool-specific
|
|
||||||
tool_name: Optional[str]
|
|
||||||
was_recommended: Optional[bool]
|
|
||||||
was_actually_used: Optional[bool]
|
|
||||||
|
|
||||||
# Context
|
|
||||||
conversation_id: Optional[str]
|
|
||||||
metadata: dict
|
|
||||||
```
|
|
||||||
|
|
||||||
**1.5 Testing**
|
|
||||||
|
|
||||||
- Test household registry registration and retrieval
|
|
||||||
- Test Toolset composition
|
|
||||||
- Test benchmark recording to Redis
|
|
||||||
- Test structured logging output
|
|
||||||
|
|
||||||
#### Success Criteria
|
|
||||||
- ✅ Household registry operational
|
|
||||||
- ✅ Tatlock core tools organized in domain directory
|
|
||||||
- ✅ Redis benchmarks working
|
|
||||||
- ✅ Structured logging functional
|
|
||||||
- ✅ Tests pass and maintain 80%+ coverage
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
|
|
||||||
### Milestone 2: Minimal Steward Agent with Context Analysis (Week 3-4)
|
|
||||||
|
|
||||||
#### Goal
|
|
||||||
Create a Steward agent that analyzes requests with full conversation context and recommends relevant household capabilities.
|
|
||||||
|
|
||||||
#### Tasks
|
|
||||||
|
|
||||||
**2.1 Create Steward Agent**
|
|
||||||
|
|
||||||
Location: `src/agents/steward/agent.py`
|
|
||||||
|
|
||||||
Structured output schema:
|
|
||||||
```python
|
|
||||||
class ConversationContext(BaseModel):
|
|
||||||
"""Contextual information from conversation history."""
|
|
||||||
has_previous_context: bool
|
|
||||||
relevant_turns: list[int] # 0-indexed turn numbers
|
|
||||||
context_summary: str # Summary for Butler
|
|
||||||
|
|
||||||
class StewardRecommendation(BaseModel):
|
|
||||||
"""Structured recommendation from Steward analysis."""
|
|
||||||
recommended_capabilities: list[str]
|
|
||||||
reasoning: str
|
|
||||||
estimated_complexity: Literal["simple", "moderate", "complex"]
|
|
||||||
conversation_context: ConversationContext
|
|
||||||
missing_capabilities: Optional[str] = None
|
|
||||||
```
|
|
||||||
|
|
||||||
Key features:
|
|
||||||
- Uses same model as Tatlock (`ollama:mistral-nemo`) for VRAM efficiency
|
|
||||||
- Receives FULL conversation history
|
|
||||||
- Queries household registry via tool
|
|
||||||
- Conservative recommendations (avoid over-inclusion)
|
|
||||||
- Explicit handling of missing capabilities
|
|
||||||
|
|
||||||
**2.2 Steward System Prompt**
|
|
||||||
|
|
||||||
Responsibilities:
|
|
||||||
1. **Capability Recommendation**: Query registry, recommend only necessary tools
|
|
||||||
2. **Conversation Analysis**: Identify references to previous topics
|
|
||||||
3. **Complexity Assessment**: Simple/moderate/complex classification
|
|
||||||
4. **Missing Capability Detection**: Suggest what's needed if no tools available
|
|
||||||
|
|
||||||
**2.3 Steward Service Layer with Logging**
|
|
||||||
|
|
||||||
Location: `src/agents/steward/service.py`
|
|
||||||
|
|
||||||
```python
|
|
||||||
async def analyze_request(
|
|
||||||
user_request: str,
|
|
||||||
conversation_history: list[dict] # FULL conversation
|
|
||||||
) -> StewardRecommendation:
|
|
||||||
"""Analyze request with full conversation context."""
|
|
||||||
|
|
||||||
async with log_operation("steward_analysis", {...}) as log_ctx:
|
|
||||||
result = await steward_agent.run(
|
|
||||||
user_request,
|
|
||||||
message_history=convert_to_pydantic_history(conversation_history),
|
|
||||||
usage_limits=UsageLimits(request_limit=3)
|
|
||||||
)
|
|
||||||
|
|
||||||
# Log and benchmark
|
|
||||||
log_ctx["recommendation_count"] = len(result.data.recommended_capabilities)
|
|
||||||
await benchmark_store.record(...)
|
|
||||||
|
|
||||||
return result.data
|
|
||||||
```
|
|
||||||
|
|
||||||
**2.4 Testing**
|
|
||||||
|
|
||||||
Test scenarios:
|
|
||||||
- Calculator request → recommends tatlock_core
|
|
||||||
- Simple greeting → recommends []
|
|
||||||
- Web search request → recommends tatlock_core
|
|
||||||
- Request referencing previous turn → identifies context
|
|
||||||
- Impossible request → returns missing_capabilities
|
|
||||||
|
|
||||||
#### Success Criteria
|
|
||||||
- ✅ Steward queries household registry successfully
|
|
||||||
- ✅ Produces structured recommendations
|
|
||||||
- ✅ Analyzes full conversation context
|
|
||||||
- ✅ Handles missing capabilities gracefully
|
|
||||||
- ✅ Conservative recommendations (> 90% accuracy)
|
|
||||||
- ✅ Benchmarks recorded to Redis
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
|
|
||||||
### Milestone 3: Request Preprocessing & Tool Tracking (Week 5-6)
|
|
||||||
|
|
||||||
#### Goal
|
|
||||||
Wire Steward into request flow, implement tool scoping, and track tool usage.
|
|
||||||
|
|
||||||
#### Tasks
|
|
||||||
|
|
||||||
**3.1 Create Preprocessing Pipeline**
|
|
||||||
|
|
||||||
Location: `src/core/preprocessing.py`
|
|
||||||
|
|
||||||
```python
|
|
||||||
@dataclass
|
|
||||||
class EnrichedRequest:
|
|
||||||
"""Request enriched with Steward's analysis."""
|
|
||||||
original_request: str
|
|
||||||
steward_note: str # Formatted note for Tatlock
|
|
||||||
scoped_toolset: CombinedToolset # Only recommended tools
|
|
||||||
recommendation: StewardRecommendation
|
|
||||||
steward_reasoning_output: str # For streaming to user
|
|
||||||
|
|
||||||
async def preprocess_request(
|
|
||||||
user_request: str,
|
|
||||||
conversation_history: list[dict] # FULL conversation
|
|
||||||
) -> EnrichedRequest:
|
|
||||||
"""Analyze via Steward and prepare scoped context."""
|
|
||||||
# Call Steward with full conversation
|
|
||||||
recommendation = await analyze_request(user_request, conversation_history)
|
|
||||||
|
|
||||||
# Format note to Tatlock (includes conversation context)
|
|
||||||
steward_note = format_steward_note(recommendation)
|
|
||||||
|
|
||||||
# Create scoped toolset
|
|
||||||
scoped_toolset = household_registry.get_scoped_toolset(
|
|
||||||
recommendation.recommended_capabilities
|
|
||||||
)
|
|
||||||
|
|
||||||
return EnrichedRequest(...)
|
|
||||||
```
|
|
||||||
|
|
||||||
Note formatting:
|
|
||||||
- Includes conversation context summary
|
|
||||||
- Highlights missing capabilities if applicable
|
|
||||||
- Provides complexity estimate
|
|
||||||
|
|
||||||
**3.2 Tool Usage Tracking**
|
|
||||||
|
|
||||||
Location: `src/core/tool_tracking.py`
|
|
||||||
|
|
||||||
```python
|
|
||||||
class ToolCallTracker:
|
|
||||||
"""Tracks tool calls for benchmarking."""
|
|
||||||
|
|
||||||
def __init__(self, recommended_tools: list[str]):
|
|
||||||
self.recommended_tools = set(recommended_tools)
|
|
||||||
self.actual_calls: dict[str, list[float]] = {}
|
|
||||||
|
|
||||||
async def track_call(self, tool_name: str, duration: float):
|
|
||||||
"""Record a tool call with timing."""
|
|
||||||
# Log if tool wasn't recommended
|
|
||||||
if tool_name not in self.recommended_tools:
|
|
||||||
logger.warning("tool_call_not_recommended", ...)
|
|
||||||
|
|
||||||
# Record benchmark to Redis
|
|
||||||
await benchmark_store.record(...)
|
|
||||||
|
|
||||||
async def finalize(self):
|
|
||||||
"""Log unused recommended tools."""
|
|
||||||
unused = self.recommended_tools - set(self.actual_calls.keys())
|
|
||||||
# Record benchmarks for unused tools
|
|
||||||
```
|
|
||||||
|
|
||||||
**3.3 Integrate with Responses API**
|
|
||||||
|
|
||||||
Modify `src/responses/service.py`:
|
|
||||||
```python
|
|
||||||
async def generate_response(request: ResponseRequest) -> ResponseOutput:
|
|
||||||
# Preprocess via Steward (with full conversation)
|
|
||||||
enriched = await preprocess_request(
|
|
||||||
user_message,
|
|
||||||
conversation_history=request.input[:-1]
|
|
||||||
)
|
|
||||||
|
|
||||||
# Run Tatlock with scoped tools and tracker
|
|
||||||
result = await run_tatlock_with_scoped_tools(
|
|
||||||
enriched.original_request,
|
|
||||||
enriched.steward_note,
|
|
||||||
enriched.scoped_toolset,
|
|
||||||
enriched.recommendation.recommended_capabilities, # For tracking
|
|
||||||
message_history,
|
|
||||||
usage_tracker
|
|
||||||
)
|
|
||||||
|
|
||||||
# Build response with Steward reasoning
|
|
||||||
return build_response_with_steward_reasoning(...)
|
|
||||||
```
|
|
||||||
|
|
||||||
**3.4 Update Tatlock Agent**
|
|
||||||
|
|
||||||
Location: `src/agents/tatlock.py`
|
|
||||||
|
|
||||||
```python
|
|
||||||
async def run_tatlock_with_scoped_tools(
|
|
||||||
user_request: str,
|
|
||||||
steward_note: str,
|
|
||||||
scoped_toolset: CombinedToolset,
|
|
||||||
recommended_tools: list[str],
|
|
||||||
message_history: list[dict],
|
|
||||||
usage: UsageeLimits
|
|
||||||
):
|
|
||||||
# Initialize tracker
|
|
||||||
tracker = ToolCallTracker(recommended_tools)
|
|
||||||
|
|
||||||
# Prepend Steward's note (invisible to user, visible to Tatlock)
|
|
||||||
enriched_prompt = f"{steward_note}\n\n{user_request}"
|
|
||||||
|
|
||||||
# Run with ONLY scoped tools
|
|
||||||
result = await tatlock_agent.run(
|
|
||||||
enriched_prompt,
|
|
||||||
message_history=convert_to_pydantic_history(message_history),
|
|
||||||
toolsets=[scoped_toolset], # Tool scoping enforced
|
|
||||||
deps=tracker, # For tracking
|
|
||||||
usage=usage
|
|
||||||
)
|
|
||||||
|
|
||||||
# Finalize tracking
|
|
||||||
await tracker.finalize()
|
|
||||||
|
|
||||||
return result
|
|
||||||
```
|
|
||||||
|
|
||||||
**3.5 Add Streaming Transparency**
|
|
||||||
|
|
||||||
Modify `src/responses/streaming.py`:
|
|
||||||
- Stream Steward's reasoning first
|
|
||||||
- Then stream Tatlock's response
|
|
||||||
- Include conversation context notes
|
|
||||||
- Format missing capabilities warnings
|
|
||||||
|
|
||||||
**3.6 Testing**
|
|
||||||
|
|
||||||
Integration tests:
|
|
||||||
- Full Steward → Tatlock flow
|
|
||||||
- Tool scoping enforcement (can't use non-recommended tools)
|
|
||||||
- Tool usage tracking (recommended vs. actual)
|
|
||||||
- Conversation context propagation
|
|
||||||
- Missing capabilities handling
|
|
||||||
|
|
||||||
#### Success Criteria
|
|
||||||
- ✅ Full request flow working (User → Steward → Tatlock)
|
|
||||||
- ✅ Steward reasoning visible in output stream
|
|
||||||
- ✅ Tool scoping enforced (only recommended tools available)
|
|
||||||
- ✅ Tool usage tracked and logged to Redis
|
|
||||||
- ✅ Conversation context passed through pipeline
|
|
||||||
- ✅ Integration tests pass end-to-end
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
|
|
||||||
### Milestone 4: Testing, Benchmarking & Refinement (Week 7)
|
|
||||||
|
|
||||||
#### Goal
|
|
||||||
Validate the system, optimize performance, refine prompts, and establish monitoring.
|
|
||||||
|
|
||||||
#### Tasks
|
|
||||||
|
|
||||||
**4.1 Comprehensive Testing**
|
|
||||||
|
|
||||||
Test categories:
|
|
||||||
- End-to-end integration tests (full request flow)
|
|
||||||
- Performance benchmarks (latency targets)
|
|
||||||
- Prompt refinement (recommendation accuracy)
|
|
||||||
- Edge cases (errors, timeouts, missing capabilities)
|
|
||||||
- Conversation context accuracy
|
|
||||||
|
|
||||||
**4.2 Performance Validation**
|
|
||||||
|
|
||||||
Targets:
|
|
||||||
- Steward analysis: < 2 seconds
|
|
||||||
- Total added latency: < 3 seconds
|
|
||||||
- Model stays hot in VRAM (no reload delays)
|
|
||||||
- Tool recommendation accuracy: > 90%
|
|
||||||
|
|
||||||
**4.3 Benchmark Analysis Tools**
|
|
||||||
|
|
||||||
Create `scripts/benchmark_analysis.py`:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# View Steward performance over last 24 hours
|
|
||||||
python scripts/benchmark_analysis.py --operation steward_analysis --hours 24
|
|
||||||
|
|
||||||
# Analyze tool recommendation accuracy
|
|
||||||
python scripts/benchmark_analysis.py --tool-accuracy --days 7
|
|
||||||
```
|
|
||||||
|
|
||||||
Metrics to track:
|
|
||||||
- Average Steward analysis time
|
|
||||||
- Recommendation count distribution
|
|
||||||
- Tool accuracy (recommended & used, recommended but unused, not recommended but used)
|
|
||||||
- Recommendation precision percentage
|
|
||||||
|
|
||||||
**4.4 Prompt Engineering**
|
|
||||||
|
|
||||||
Iterate on Steward system prompt:
|
|
||||||
- Test with diverse request types
|
|
||||||
- Tune conservativeness (balance false positives/negatives)
|
|
||||||
- Validate conversation context analysis
|
|
||||||
- Test missing capability detection
|
|
||||||
|
|
||||||
**4.5 Documentation**
|
|
||||||
|
|
||||||
Update documentation:
|
|
||||||
- README.md: Steward explanation and examples
|
|
||||||
- AGENTS.md: Household registration pattern
|
|
||||||
- IMPLEMENTATION_ROADMAP.md: Mark Phase 2 complete
|
|
||||||
- Add benchmark analysis guide
|
|
||||||
|
|
||||||
#### Success Criteria
|
|
||||||
- ✅ < 3 seconds added latency for Steward analysis
|
|
||||||
- ✅ > 90% recommendation accuracy (manual evaluation)
|
|
||||||
- ✅ All integration tests pass
|
|
||||||
- ✅ Benchmark tools functional
|
|
||||||
- ✅ Documentation complete and accurate
|
|
||||||
- ✅ Ready for Phase 3/4 (expert agents)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Architecture Diagram
|
|
||||||
|
|
||||||
```
|
|
||||||
User Request
|
|
||||||
↓
|
|
||||||
Orchestrator (FastAPI)
|
|
||||||
↓
|
|
||||||
Preprocessing Pipeline
|
|
||||||
├─→ Steward Agent
|
|
||||||
│ ├─ Receives: FULL conversation history
|
|
||||||
│ ├─ Analyzes: Context, references, requirements
|
|
||||||
│ ├─ Queries: Household registry (capabilities)
|
|
||||||
│ ├─ Outputs: StewardRecommendation
|
|
||||||
│ │ ├─ recommended_capabilities: list[str]
|
|
||||||
│ │ ├─ conversation_context: ConversationContext
|
|
||||||
│ │ ├─ missing_capabilities: str | None
|
|
||||||
│ │ └─ reasoning: str
|
|
||||||
│ └─ Logs: Performance benchmarks → Redis
|
|
||||||
│
|
|
||||||
├─→ Create Scoped Toolset
|
|
||||||
│ └─ CombinedToolset from recommended capabilities
|
|
||||||
│
|
|
||||||
└─→ Format Steward Note
|
|
||||||
└─ Includes conversation context for Tatlock
|
|
||||||
↓
|
|
||||||
Tatlock Agent (with scoped tools)
|
|
||||||
├─ Receives: Enriched request + Steward note
|
|
||||||
├─ Has access to: ONLY recommended tools
|
|
||||||
├─ Tool calls tracked: ToolCallTracker
|
|
||||||
└─ Logs: Tool usage benchmarks → Redis
|
|
||||||
↓
|
|
||||||
Response to User
|
|
||||||
├─ Steward's reasoning (streamed first)
|
|
||||||
└─ Tatlock's response (streamed second)
|
|
||||||
|
|
||||||
Background:
|
|
||||||
└─ Redis: Performance benchmarks, tool usage analysis
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Design Decisions Summary
|
|
||||||
|
|
||||||
### 1. Logging & Performance Benchmarks
|
|
||||||
**Decision**: Full observability with Redis-backed benchmark storage
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Track Steward recommendations vs. Tatlock's actual tool usage
|
|
||||||
- Measure performance metrics (latency, token usage)
|
|
||||||
- Cross-session analysis for optimization
|
|
||||||
- Identify recommendation accuracy over time
|
|
||||||
|
|
||||||
### 2. Steward Fallback Behavior
|
|
||||||
**Decision**: Explicit missing capability communication
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- No suitable tools → Steward states "missing capabilities" with description
|
|
||||||
- Can suggest what type of tool would be helpful
|
|
||||||
- Code errors → standard exception handlers (don't suppress real errors)
|
|
||||||
- Better UX than silent failures or defaulting to all tools
|
|
||||||
|
|
||||||
### 3. Conversation History for Steward
|
|
||||||
**Decision**: Steward sees FULL conversation, not just current turn
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Can identify references to previous topics
|
|
||||||
- Provides contextual notes to Butler
|
|
||||||
- "Two sets of eyes" on conversation
|
|
||||||
- Example: "User mentioned Python debugging in turn 3, relevant details: async code"
|
|
||||||
|
|
||||||
### 4. Registry Pattern
|
|
||||||
**Decision**: Separate Household Registry from Model Registry
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Tools belong to household members, not models
|
|
||||||
- Clean separation of concerns
|
|
||||||
- Executive summaries for coordination, details for execution
|
|
||||||
|
|
||||||
### 5. Tool Composition
|
|
||||||
**Decision**: PydanticAI FunctionToolset + CombinedToolset
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Native PydanticAI pattern
|
|
||||||
- Clean composition and filtering
|
|
||||||
- Dynamic scoping per request
|
|
||||||
|
|
||||||
### 6. Tool Scoping
|
|
||||||
**Decision**: Compile-time scoping via toolset creation
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Tools not even visible to LLM
|
|
||||||
- Cleaner than runtime permission checks
|
|
||||||
- Enforced at PydanticAI level
|
|
||||||
|
|
||||||
### 7. Organization
|
|
||||||
**Decision**: Domain-based household directories
|
|
||||||
|
|
||||||
**Rationale**:
|
|
||||||
- Each household member owns their tools
|
|
||||||
- Clear bounded contexts
|
|
||||||
- Example: `src/agents/tatlock_core/`, `src/agents/librarian/` (future)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Infrastructure Requirements
|
|
||||||
|
|
||||||
### Redis Setup
|
|
||||||
|
|
||||||
Development (quick start):
|
|
||||||
```bash
|
|
||||||
# Docker (recommended)
|
|
||||||
docker run -d -p 6379:6379 --name tatlock-redis redis:7-alpine
|
|
||||||
|
|
||||||
# Or local installation
|
|
||||||
# macOS: brew install redis && brew services start redis
|
|
||||||
# Linux: sudo apt install redis-server && sudo systemctl start redis
|
|
||||||
```
|
|
||||||
|
|
||||||
Production (docker-compose.yml):
|
|
||||||
```yaml
|
|
||||||
services:
|
|
||||||
redis:
|
|
||||||
image: redis:7-alpine
|
|
||||||
ports:
|
|
||||||
- "6379:6379"
|
|
||||||
volumes:
|
|
||||||
- redis_data:/data
|
|
||||||
command: redis-server --appendonly yes
|
|
||||||
|
|
||||||
volumes:
|
|
||||||
redis_data:
|
|
||||||
```
|
|
||||||
|
|
||||||
### Dependencies Update
|
|
||||||
|
|
||||||
Add to `requirements.txt`:
|
|
||||||
```txt
|
|
||||||
redis[hiredis]>=5.0.0,<6.0.0
|
|
||||||
structlog>=24.1.0,<25.0.0
|
|
||||||
```
|
|
||||||
|
|
||||||
### Configuration
|
|
||||||
|
|
||||||
Add to `.env`:
|
|
||||||
```env
|
|
||||||
# Redis Configuration
|
|
||||||
REDIS_URL=redis://localhost:6379/1
|
|
||||||
|
|
||||||
# Logging
|
|
||||||
LOG_LEVEL=INFO
|
|
||||||
LOG_FORMAT=json
|
|
||||||
ENABLE_BENCHMARKS=true
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Timeline
|
|
||||||
|
|
||||||
**Week 1-2**: Household Registry + Logging Infrastructure
|
|
||||||
- Household registry with Toolsets
|
|
||||||
- Structured logging with structlog
|
|
||||||
- Redis benchmark storage
|
|
||||||
- Tatlock core reorganization
|
|
||||||
- Tests: Registry + benchmarking
|
|
||||||
|
|
||||||
**Week 3-4**: Steward Agent with Context Analysis
|
|
||||||
- Steward agent with conversation context
|
|
||||||
- ConversationContext in recommendations
|
|
||||||
- Missing capabilities handling
|
|
||||||
- Tests: Context analysis, missing capabilities
|
|
||||||
|
|
||||||
**Week 5-6**: Integration + Tool Tracking
|
|
||||||
- Request preprocessing with full conversation
|
|
||||||
- Tool usage tracking middleware
|
|
||||||
- Scoped toolset creation
|
|
||||||
- Streaming transparency
|
|
||||||
- Tests: Full flow + tool tracking
|
|
||||||
|
|
||||||
**Week 7**: Testing, Benchmarking & Refinement
|
|
||||||
- End-to-end integration tests
|
|
||||||
- Benchmark analysis tools
|
|
||||||
- Prompt refinement
|
|
||||||
- Performance validation
|
|
||||||
- Documentation updates
|
|
||||||
|
|
||||||
**Total: 4-5 weeks** (core implementation complete in 6 weeks, polish in week 7)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Success Metrics
|
|
||||||
|
|
||||||
### Technical
|
|
||||||
- ✅ Household registry operational with executive summaries
|
|
||||||
- ✅ Steward produces accurate recommendations (> 90%)
|
|
||||||
- ✅ Steward analyzes full conversation context
|
|
||||||
- ✅ Tool scoping enforced (Tatlock can't use non-recommended tools)
|
|
||||||
- ✅ Model efficiency preserved (no reload delays)
|
|
||||||
- ✅ Added latency < 3 seconds
|
|
||||||
- ✅ Performance benchmarks recorded to Redis
|
|
||||||
- ✅ Tool usage tracking (recommended vs. actual)
|
|
||||||
|
|
||||||
### Observability
|
|
||||||
- ✅ Structured logging (JSON format)
|
|
||||||
- ✅ Benchmark analysis tools available
|
|
||||||
- ✅ Tool recommendation accuracy measurable
|
|
||||||
- ✅ Cross-session performance trends visible
|
|
||||||
|
|
||||||
### Error Handling
|
|
||||||
- ✅ Missing capabilities explicitly communicated
|
|
||||||
- ✅ Steward can guide user toward needed resources
|
|
||||||
- ✅ Code errors properly surfaced (not suppressed)
|
|
||||||
|
|
||||||
### Architectural
|
|
||||||
- ✅ PydanticAI patterns followed (Toolsets, decorators, structured outputs)
|
|
||||||
- ✅ Clean separation: registry vs. agents vs. tools
|
|
||||||
- ✅ Two-tier abstraction working (summaries vs. details)
|
|
||||||
- ✅ Future-proof for expert agents (Phase 4)
|
|
||||||
|
|
||||||
### Testing
|
|
||||||
- ✅ Maintain 80%+ test coverage
|
|
||||||
- ✅ Integration tests for full flow
|
|
||||||
- ✅ Performance benchmarks established
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Future-Proofing for Phase 4
|
|
||||||
|
|
||||||
### Expert Agent Pattern (Template)
|
|
||||||
|
|
||||||
When adding The Librarian, The Developer, etc., follow this structure:
|
|
||||||
|
|
||||||
```
|
|
||||||
src/agents/librarian/
|
|
||||||
├── __init__.py
|
|
||||||
├── agent.py # Librarian PydanticAI agent
|
|
||||||
├── tools.py # Librarian-specific tools (wiki, research, etc.)
|
|
||||||
├── toolset.py # PydanticAI toolset creation
|
|
||||||
└── capability.py # Executive summary for registry
|
|
||||||
```
|
|
||||||
|
|
||||||
Example capability registration:
|
|
||||||
```python
|
|
||||||
# capability.py
|
|
||||||
LIBRARIAN_CAPABILITY = HouseholdCapability(
|
|
||||||
name="librarian",
|
|
||||||
role="The Librarian",
|
|
||||||
category="research",
|
|
||||||
description="Research assistance, knowledge management, and information synthesis",
|
|
||||||
domains=["research", "knowledge_base", "documentation"],
|
|
||||||
cost="medium",
|
|
||||||
requires_network=True
|
|
||||||
)
|
|
||||||
|
|
||||||
def register_librarian():
|
|
||||||
household_registry.register(
|
|
||||||
name="librarian",
|
|
||||||
capability=LIBRARIAN_CAPABILITY,
|
|
||||||
toolset=librarian_toolset,
|
|
||||||
agent=librarian_agent # Expert agent for delegation
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
Tatlock delegation pattern (Phase 4):
|
|
||||||
```python
|
|
||||||
@tatlock_agent.tool
|
|
||||||
async def consult_librarian(
|
|
||||||
ctx: RunContext[None],
|
|
||||||
research_query: str
|
|
||||||
) -> str:
|
|
||||||
"""Consult the Librarian for research assistance."""
|
|
||||||
from src.agents.librarian.agent import librarian_agent
|
|
||||||
|
|
||||||
result = await librarian_agent.run(
|
|
||||||
research_query,
|
|
||||||
usage=ctx.usage # Aggregate usage
|
|
||||||
)
|
|
||||||
return result.data
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Risk Mitigation
|
|
||||||
|
|
||||||
### Identified Risks
|
|
||||||
|
|
||||||
1. **Steward recommendations too broad**
|
|
||||||
- Mitigation: Conservative prompt engineering, benchmark tracking, iterate based on false positives
|
|
||||||
|
|
||||||
2. **Added latency unacceptable**
|
|
||||||
- Mitigation: Stream Steward reasoning for transparency, optimize prompt, use same base model
|
|
||||||
|
|
||||||
3. **Tool registry becomes unwieldy**
|
|
||||||
- Mitigation: Good categorization, semantic search (future), regular pruning
|
|
||||||
|
|
||||||
4. **Model VRAM competition**
|
|
||||||
- Mitigation: Use same base model for Steward and Tatlock, sequential calls
|
|
||||||
|
|
||||||
5. **Redis dependency**
|
|
||||||
- Mitigation: Make benchmarking optional, graceful degradation if Redis unavailable
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Open Questions - RESOLVED
|
|
||||||
|
|
||||||
All major design questions have been resolved. See "Design Decisions Summary" section above.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Next Steps
|
|
||||||
|
|
||||||
### Immediate (Today/This Week)
|
|
||||||
1. Set up Redis (Docker or local)
|
|
||||||
2. Create `src/core/logging_config.py` with structured logging
|
|
||||||
3. Create `src/core/benchmarks.py` with Redis storage
|
|
||||||
4. Add `redis` and `structlog` to requirements.txt
|
|
||||||
5. Create household registry skeleton
|
|
||||||
|
|
||||||
### Week 1-2
|
|
||||||
1. Complete household registry with Toolset integration
|
|
||||||
2. Reorganize Tatlock core tools into domain directory
|
|
||||||
3. Implement logging infrastructure
|
|
||||||
4. Write tests for registry + benchmarking
|
|
||||||
|
|
||||||
### Week 3-4
|
|
||||||
1. Create Steward agent with conversation context
|
|
||||||
2. Implement missing capabilities handling
|
|
||||||
3. Test context analysis accuracy
|
|
||||||
4. Iterate on system prompt
|
|
||||||
|
|
||||||
### Week 5-6
|
|
||||||
1. Build preprocessing pipeline
|
|
||||||
2. Integrate with Responses API
|
|
||||||
3. Implement tool tracking
|
|
||||||
4. Add streaming transparency
|
|
||||||
|
|
||||||
### Week 7
|
|
||||||
1. End-to-end testing
|
|
||||||
2. Benchmark analysis
|
|
||||||
3. Performance optimization
|
|
||||||
4. Documentation updates
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Document Status
|
|
||||||
|
|
||||||
**Status**: Active Planning Document
|
|
||||||
**Created**: 2025-12-07
|
|
||||||
**Last Updated**: 2025-12-07
|
|
||||||
**Version**: 1.0
|
|
||||||
**Next Review**: After Milestone 1 completion
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Reference Documents**:
|
|
||||||
- [PHILOSOPHY.md](PHILOSOPHY.md) - System vision and architecture
|
|
||||||
- [IMPLEMENTATION_ROADMAP.md](IMPLEMENTATION_ROADMAP.md) - Full project roadmap
|
|
||||||
- [AGENTS.md](AGENTS.md) - Agent development guidelines
|
|
||||||
- [README.md](README.md) - User documentation
|
|
||||||
|
|
||||||
@@ -1,17 +1,30 @@
|
|||||||
# Tatlock - Your Homelab Butler
|
# Tatlock - Your Homelab Butler
|
||||||
|
|
||||||
> **📖 For the complete system vision and architectural philosophy, see [PHILOSOPHY.md](PHILOSOPHY.md)**
|
> **📖 For the complete system vision and architectural philosophy, see [docs/philosophy.md](docs/philosophy.md)**
|
||||||
|
|
||||||
A privacy-first, offline-capable personal assistant system that coordinates specialized AI agents to help with research, development, home automation, and daily organization.
|
A privacy-first, offline-capable personal assistant system that coordinates specialized AI agents to help with research, development, home automation, and daily organization.
|
||||||
|
|
||||||
## Current Status
|
## Current Status
|
||||||
|
|
||||||
- ✅ **Production-ready testing API** with OpenAI Responses API format
|
- ✅ **Production-ready API** with OpenAI Responses API format
|
||||||
- ✅ **Open WebUI integration** with reasoning bubbles (`<think>` tags)
|
- ✅ **Open WebUI integration** with reasoning bubbles (`<think>` tags)
|
||||||
- ✅ **Conversation history** with auto-generated IDs and context management
|
- ✅ **Two-tier architecture** - The Steward analyzes requests, Tatlock coordinates execution
|
||||||
- ✅ **Tatlock PydanticAI Agent** - Real LLM integration with Ollama + permanent tools
|
- ✅ **Multi-agent coordination** - Expert household staff for specialized tasks
|
||||||
- ✅ **Permanent Tools** - Calculator, date/time toolkit, web search (SearXNG)
|
- ✅ **Memory system** - User profile, preferences, and semantic recall
|
||||||
- ✅ **Comprehensive testing** - 131 tests, 81.78% coverage
|
- ✅ **Comprehensive testing** - 399 tests with good coverage
|
||||||
|
|
||||||
|
### The Household Staff
|
||||||
|
|
||||||
|
| Agent | Role | Status |
|
||||||
|
|-------|------|--------|
|
||||||
|
| **Tatlock** | The Butler - Primary interface with witty personality | ✅ Active |
|
||||||
|
| **The Steward** | Request analysis and capability recommendation | ✅ Active |
|
||||||
|
| **The Librarian** | Research, wiki management, knowledge synthesis | ✅ Active |
|
||||||
|
| **The Biographer** | User memory - profiles, preferences, facts | ✅ Active |
|
||||||
|
| **The Developer** | Code assistance, debugging, architecture | 🔜 Planned |
|
||||||
|
| **The Secretary** | Scheduling, calendars, reminders | 🔜 Planned |
|
||||||
|
| **The Handyman** | System administration, monitoring | 🔜 Planned |
|
||||||
|
| **The Housekeeper** | Home automation (Home Assistant) | 🔜 Planned |
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
@@ -45,24 +58,27 @@ A privacy-first, offline-capable personal assistant system that coordinates spec
|
|||||||
- Error triggers for testing (rate_limit, context_overflow)
|
- Error triggers for testing (rate_limit, context_overflow)
|
||||||
|
|
||||||
- **Tatlock**: Real PydanticAI agent with butler personality
|
- **Tatlock**: Real PydanticAI agent with butler personality
|
||||||
- **LLM Backend**: Ollama (mistral-nemo:latest)
|
- **LLM Backend**: Ollama (gemma4:e2b by default, local-first) with optional Claude fallback
|
||||||
- **Personality**: Witty British butler, research-oriented
|
- **Personality**: Witty British butler, research-oriented
|
||||||
- **Permanent Tools**:
|
- **Core Tools**:
|
||||||
- **Calculator**: Safe mathematical expression evaluation (arithmetic, algebra, trigonometry, logarithms)
|
- **Calculator**: Safe mathematical expression evaluation
|
||||||
- **Date/Time Toolkit**: Current time, relative dates ("1 week ago"), time differences
|
- **Date/Time Toolkit**: Current time, relative dates, time differences
|
||||||
- **Web Search**: Privacy-preserving search via SearXNG
|
- **Web Search**: Privacy-preserving search via SearXNG
|
||||||
- **Capabilities**: Streaming, reasoning, tool calling
|
- **Household Coordination**:
|
||||||
- **Phase**: Phase 1 - Basic Integration (full household coordination coming in future phases)
|
- **The Steward**: Analyzes requests and recommends capabilities
|
||||||
|
- **The Librarian**: Research via library-desk HybridRAG + wiki
|
||||||
|
- **The Biographer**: User memory and preference management
|
||||||
|
- **Capabilities**: Streaming, reasoning, tool calling, multi-agent delegation
|
||||||
|
|
||||||
## Requirements
|
## Requirements
|
||||||
|
|
||||||
- Python 3.12+ (Python 3.12.11 recommended)
|
- Python 3.12+ (Python 3.12.11 recommended)
|
||||||
- **Ollama** (for Tatlock agent): Running locally or network-accessible
|
- **External Services** (must be running separately):
|
||||||
- Download: https://ollama.ai/
|
- **Ollama**: LLM inference (gemma4:e2b, nomic-embed-text)
|
||||||
- Model: `ollama pull mistral-nemo:latest`
|
- **Redis**: Caching and session memory
|
||||||
- **SearXNG** (for web search tool): Optional but recommended
|
- **Qdrant**: Vector storage for The Biographer's memory
|
||||||
- Docker: `docker run -d -p 8087:8080 searxng/searxng`
|
- **SearXNG**: Web search (optional)
|
||||||
- Or use public instance (less private)
|
- **library-desk**: Research API for The Librarian (optional)
|
||||||
|
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
@@ -73,12 +89,8 @@ A privacy-first, offline-capable personal assistant system that coordinates spec
|
|||||||
git clone https://git.schweitz.net/jpmschweitzer/tatlock.git
|
git clone https://git.schweitz.net/jpmschweitzer/tatlock.git
|
||||||
cd tatlock
|
cd tatlock
|
||||||
|
|
||||||
# Create virtual environment
|
|
||||||
python -m venv .venv
|
|
||||||
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
||||||
|
|
||||||
# Install dependencies
|
# Install dependencies
|
||||||
pip install -r requirements.txt
|
make setup
|
||||||
```
|
```
|
||||||
|
|
||||||
### Run the Server
|
### Run the Server
|
||||||
@@ -251,15 +263,21 @@ Interactive documentation available at:
|
|||||||
# Run all tests
|
# Run all tests
|
||||||
pytest
|
pytest
|
||||||
|
|
||||||
|
# Run unit tests only (no external services needed)
|
||||||
|
pytest --ignore=tests/e2e --ignore=tests/integration --ignore=tests/contracts
|
||||||
|
|
||||||
|
# Wire-level contract tests against live service boundaries
|
||||||
|
make test-contracts
|
||||||
|
|
||||||
# Run with coverage
|
# Run with coverage
|
||||||
pytest --cov=src --cov-report=term-missing
|
pytest --cov=src --cov-report=term-missing
|
||||||
|
|
||||||
# Current: 131 tests, 81.78% coverage
|
# Current: ~400 tests
|
||||||
```
|
```
|
||||||
|
|
||||||
**Test Categories:**
|
**Test Categories:**
|
||||||
- Unit tests: Agent tools, streaming, schemas
|
- Unit tests: Agent tools, capabilities, schemas, memory service
|
||||||
- Integration tests: Full API stack with real Ollama calls
|
- Integration tests: Full API stack with real Ollama
|
||||||
- End-to-end tests: Chat completions, responses API
|
- End-to-end tests: Chat completions, responses API
|
||||||
|
|
||||||
## Deployment
|
## Deployment
|
||||||
@@ -288,12 +306,33 @@ Create a `.env` file for custom configuration:
|
|||||||
API_HOST=0.0.0.0
|
API_HOST=0.0.0.0
|
||||||
API_PORT=8000
|
API_PORT=8000
|
||||||
|
|
||||||
# Ollama Configuration
|
# Ollama Configuration (primary backend)
|
||||||
OLLAMA_HOST=http://localhost:11434
|
OLLAMA_HOST=http://localhost:11434
|
||||||
OLLAMA_DEFAULT_MODEL=mistral-nemo:latest
|
OLLAMA_DEFAULT_MODEL=gemma4:e2b
|
||||||
|
OLLAMA_EMBEDDING_MODEL=nomic-embed-text
|
||||||
OLLAMA_TIMEOUT=120
|
OLLAMA_TIMEOUT=120
|
||||||
|
|
||||||
# SearXNG Configuration (for web search tool)
|
# Claude fallback (optional; used when Ollama is down or PREFER_CLOUD_BACKEND=true)
|
||||||
|
# ANTHROPIC_API_KEY=sk-ant-api03-your-key-here
|
||||||
|
ANTHROPIC_MODEL=claude-sonnet-5
|
||||||
|
PREFER_CLOUD_BACKEND=false
|
||||||
|
|
||||||
|
# Redis Configuration
|
||||||
|
REDIS_HOST=localhost
|
||||||
|
REDIS_PORT=6379
|
||||||
|
REDIS_MEMORY_DB=2
|
||||||
|
REDIS_MEMORY_TTL_HOURS=24
|
||||||
|
|
||||||
|
# Qdrant Configuration (for memory)
|
||||||
|
QDRANT_HOST=localhost
|
||||||
|
QDRANT_PORT=6333
|
||||||
|
QDRANT_EMBEDDING_DIM=768
|
||||||
|
|
||||||
|
# Library-desk Configuration (for The Librarian)
|
||||||
|
LIBRARY_DESK_HOST=http://localhost:8089
|
||||||
|
LIBRARY_DESK_TIMEOUT=60
|
||||||
|
|
||||||
|
# SearXNG Configuration (for web search)
|
||||||
SEARXNG_HOST=http://localhost:8087
|
SEARXNG_HOST=http://localhost:8087
|
||||||
SEARXNG_TIMEOUT=30
|
SEARXNG_TIMEOUT=30
|
||||||
|
|
||||||
@@ -339,28 +378,35 @@ See `.env.example` for full configuration options.
|
|||||||
```
|
```
|
||||||
tatlock/
|
tatlock/
|
||||||
├── src/
|
├── src/
|
||||||
│ ├── agents/ # Agent interface and implementations
|
│ ├── agents/ # Agent implementations
|
||||||
│ │ ├── base.py # AgentInterface abstract class
|
│ │ ├── biographer/ # The Biographer - memory management
|
||||||
│ │ ├── lorem_tester.py # Mock agent for testing
|
│ │ ├── librarian/ # The Librarian - research & wiki
|
||||||
│ │ ├── tatlock.py # Real PydanticAI butler agent
|
│ │ ├── steward/ # The Steward - request analysis
|
||||||
│ │ ├── tools.py # Permanent tools (calculator, date/time, search)
|
│ │ ├── tatlock_core/ # Core butler tools
|
||||||
│ │ └── registry.py # Model registry
|
│ │ ├── tatlock.py # Tatlock PydanticAI agent
|
||||||
│ ├── responses/ # Responses API (primary endpoint)
|
│ │ ├── delegation.py # Expert delegation wrappers
|
||||||
│ ├── chat/ # Chat Completions wrapper
|
│ │ └── protocol.py # Agent error protocol
|
||||||
│ ├── models/ # Models listing
|
│ ├── responses/ # Responses API (primary endpoint)
|
||||||
│ ├── core/ # Shared utilities and config
|
│ ├── chat/ # Chat Completions wrapper
|
||||||
│ └── main.py # Application entry point
|
│ ├── models/ # Models listing
|
||||||
├── tests/ # Comprehensive test suite (131 tests)
|
│ ├── core/ # Shared infrastructure
|
||||||
├── AGENTS.md # LLM agent development guidelines
|
│ │ ├── config.py # Configuration management
|
||||||
├── PHILOSOPHY.md # System vision and architecture
|
│ │ ├── context.py # Request context (ContextVar)
|
||||||
├── IMPLEMENTATION_ROADMAP.md # Development phases
|
│ │ ├── memory_service.py # Direct memory access
|
||||||
├── CHANGELOG.md # Version history
|
│ │ ├── memory_cache.py # Redis session cache
|
||||||
└── README.md # This file
|
│ │ ├── embeddings.py # Ollama embedding client
|
||||||
|
│ │ ├── qdrant.py # Vector database client
|
||||||
|
│ │ └── multi_tenancy.py # User isolation utilities
|
||||||
|
│ └── main.py # Application entry point
|
||||||
|
├── tests/ # Comprehensive test suite
|
||||||
|
├── docs/ # Project documentation
|
||||||
|
├── CHANGELOG.md # Version history
|
||||||
|
└── README.md # This file
|
||||||
```
|
```
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
For LLM agent development guidelines and architectural decisions, see [AGENTS.md](AGENTS.md).
|
For LLM agent development guidelines and architectural decisions, see [CLAUDE.md](CLAUDE.md).
|
||||||
|
|
||||||
## Contributing
|
## Contributing
|
||||||
|
|
||||||
@@ -372,9 +418,9 @@ For LLM agent development guidelines and architectural decisions, see [AGENTS.md
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
- **System Philosophy**: [PHILOSOPHY.md](PHILOSOPHY.md) - Vision, goals, and architectural patterns
|
- **System Philosophy**: [docs/philosophy.md](docs/philosophy.md) - Vision, goals, and architectural patterns
|
||||||
- **User Guide**: This file - Installation, usage, and examples
|
- **Development Roadmap**: [docs/roadmap.md](docs/roadmap.md) - Open work and planned phases
|
||||||
- **Developer Guidelines**: [AGENTS.md](AGENTS.md) - LLM agent development patterns
|
- **Developer Guidelines**: [CLAUDE.md](CLAUDE.md) - LLM agent development patterns
|
||||||
- **Version History**: [CHANGELOG.md](CHANGELOG.md) - Changes and releases
|
- **Version History**: [CHANGELOG.md](CHANGELOG.md) - Changes and releases
|
||||||
|
|
||||||
### External References
|
### External References
|
||||||
@@ -388,8 +434,8 @@ For LLM agent development guidelines and architectural decisions, see [AGENTS.md
|
|||||||
|
|
||||||
## Version
|
## Version
|
||||||
|
|
||||||
Current version: **0.2.5** - Phase 2: The Steward (Two-Tier Architecture)
|
Current version: see [CHANGELOG.md](CHANGELOG.md)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
**Note**: This is a production-ready testing API with mock responses. The architecture is designed for easy integration with real LLM backends (PydanticAI, Ollama, OpenAI, etc.).
|
**Note**: Tatlock is a production-ready homelab butler. All household staff use PydanticAI with local Ollama inference (gemma4), with an optional Claude cloud fallback.
|
||||||
|
|||||||
Executable
+50
@@ -0,0 +1,50 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Secret scan over the commits about to be pushed.
|
||||||
|
#
|
||||||
|
# Lives here rather than inside .githooks/pre-push so it can be read, run by
|
||||||
|
# hand (`make secrets`), and changed under review. A hook is a trigger; it is
|
||||||
|
# not a home for logic. Identical in every repo in this workspace (D-27).
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
cd "$(git rev-parse --show-toplevel)"
|
||||||
|
|
||||||
|
# A non-login shell — which is what git gives a hook — skips /etc/profile.d
|
||||||
|
# and never sees ~/.local/bin, where the gitleaks release tarball lands.
|
||||||
|
# Without this the scan reports "not installed" on every push.
|
||||||
|
[ -d "$HOME/.local/bin" ] && PATH="$HOME/.local/bin:$PATH"
|
||||||
|
|
||||||
|
if ! command -v gitleaks >/dev/null 2>&1; then
|
||||||
|
echo "FAIL secrets — gitleaks not installed, so this check would be a no-op pretending to pass." >&2
|
||||||
|
echo " https://github.com/gitleaks/gitleaks/releases → ~/.local/bin/gitleaks" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Scan the outgoing range, not full history. History here carries findings
|
||||||
|
# that are settled — test fixtures and vendored third-party code — and a gate
|
||||||
|
# that fails on something unfixable gets bypassed within a week. What matters
|
||||||
|
# is what is about to leave this machine.
|
||||||
|
if upstream=$(git rev-parse --abbrev-ref --symbolic-full-name '@{u}' 2>/dev/null); then
|
||||||
|
range="$upstream..HEAD"
|
||||||
|
elif git rev-parse --verify --quiet origin/main >/dev/null; then
|
||||||
|
range="origin/main..HEAD"
|
||||||
|
else
|
||||||
|
range=""
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ -z "$range" ]; then
|
||||||
|
gitleaks dir . --redact --no-banner --exit-code 1 || {
|
||||||
|
echo "FAIL secrets — gitleaks found a credential in the working tree." >&2; exit 1; }
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
[ -n "$(git log --oneline "$range" 2>/dev/null)" ] || exit 0
|
||||||
|
|
||||||
|
gitleaks git . --log-opts="$range" --redact --no-banner --exit-code 1 >/dev/null 2>&1 || {
|
||||||
|
echo "FAIL secrets — gitleaks found a credential in the commits being pushed." >&2
|
||||||
|
echo " inspect (values redacted): gitleaks git . --log-opts=\"$range\" --redact" >&2
|
||||||
|
echo " then remove and rotate it, or suppress deliberately:" >&2
|
||||||
|
echo " inline '# gitleaks:allow <reason>'" >&2
|
||||||
|
echo " or add the fingerprint to .gitleaksignore WITH a reason" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
echo " ok secrets"
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
# Claude Integration Plan
|
||||||
|
|
||||||
|
## Overview
|
||||||
|
|
||||||
|
Tatlock uses a bidirectional Claude architecture:
|
||||||
|
- **Scenario A**: Tatlock powered by Claude backend (with Ollama fallback) — **COMPLETE**, then **rolled back to local-first**: Ollama/gemma4 is primary, Claude is retained as fallback (`PREFER_CLOUD_BACKEND=false`)
|
||||||
|
- **Scenario B**: Tatlock exposed as MCP server for external Claude instances — **OPEN**
|
||||||
|
- **Scenario C**: Offline operation via Ollama — **COMPLETE**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## MCP Server (Expose Tools to Claude) — NOT STARTED
|
||||||
|
|
||||||
|
Create an MCP server that exposes Tatlock's household tools to external Claude instances.
|
||||||
|
|
||||||
|
### New Files
|
||||||
|
|
||||||
|
```
|
||||||
|
src/mcp/
|
||||||
|
├── __init__.py
|
||||||
|
├── server.py # MCP server using mcp Python SDK
|
||||||
|
├── tool_adapters.py # Convert PydanticAI tools → MCP schemas
|
||||||
|
├── auth.py # API key authentication
|
||||||
|
└── transport.py # Streamable HTTP transport
|
||||||
|
```
|
||||||
|
|
||||||
|
### Docker Stack Addition
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
tatlock-mcp:
|
||||||
|
image: git.schweitz.net/jpmschweitzer/tatlock:latest
|
||||||
|
command: ["python", "-m", "src.mcp.server"]
|
||||||
|
ports:
|
||||||
|
- "8002:8002"
|
||||||
|
environment:
|
||||||
|
- MCP_AUTH_TOKEN=${MCP_AUTH_TOKEN}
|
||||||
|
networks:
|
||||||
|
- docker-dataplane
|
||||||
|
```
|
||||||
|
|
||||||
|
### Claude Desktop Configuration
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"mcpServers": {
|
||||||
|
"tatlock": {
|
||||||
|
"command": "npx",
|
||||||
|
"args": ["mcp-remote", "https://mcp.schweitz.net/sse", "--header", "Authorization: Bearer ${MCP_AUTH_TOKEN}"]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Checklist
|
||||||
|
|
||||||
|
- [ ] Create `src/mcp/` module
|
||||||
|
- [ ] Tool adapters (PydanticAI → MCP schema)
|
||||||
|
- [ ] Authentication middleware
|
||||||
|
- [ ] Streamable HTTP transport
|
||||||
|
- [ ] Docker stack configuration
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Future Phases
|
||||||
|
|
||||||
|
- **LiteLLM Gateway** — Unified endpoint for all models, config-driven routing
|
||||||
|
- **Multi-Provider** — Add OpenAI, Vertex AI, etc.
|
||||||
|
- **Smart Routing** — Context-aware model selection, cost ceiling enforcement
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Offline Behavior
|
||||||
|
|
||||||
|
| Scenario | Behavior |
|
||||||
|
|----------|----------|
|
||||||
|
| No API key | Use Ollama exclusively |
|
||||||
|
| API unreachable | Use Ollama, log warning |
|
||||||
|
| API rate limited | Fallback to Ollama |
|
||||||
|
|
||||||
|
| Aspect | Claude | Ollama |
|
||||||
|
|--------|--------|--------|
|
||||||
|
| Context | 200k tokens | ~8k tokens |
|
||||||
|
| Latency | 1-3s (network) | 0.5-1s (local) |
|
||||||
|
| Personality | Preserved | Preserved |
|
||||||
|
| Tools | All work | All work |
|
||||||
|
| Cost | API charges | Free |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Related Repo Handovers
|
||||||
|
|
||||||
|
Handover documents created in each repo: `PROJECT_CLAUDIFICATION_HANDOVER.md`
|
||||||
|
|
||||||
|
### Open Items
|
||||||
|
|
||||||
|
- **library-desk**: Review HybridRAG response size limits, smart_create endpoint, response formats
|
||||||
|
- **core-api**: Review list_devices response format, error messages, rate limiting
|
||||||
|
- **portainer-core**: Update stack with new env vars, configure secrets, update CONTAINERS.md
|
||||||
|
- **webber**: Review content truncation limits, extraction quality
|
||||||
|
- **tatlock-ui**: Test streaming with Claude backend, conversation history, tool call display
|
||||||
@@ -0,0 +1,246 @@
|
|||||||
|
# Housekeeper Agent Optimization Findings
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
Research with Gemini identified key issues with mistral-nemo and tool calling:
|
||||||
|
- "Pre-computation Hallucination" - model answers before using tools
|
||||||
|
- High default temperature (0.7-0.8) causes wandering
|
||||||
|
- Model is "chatty and confident" - needs explicit constraints
|
||||||
|
|
||||||
|
## Key Recommendations from Gemini Research
|
||||||
|
|
||||||
|
1. **Temperature 0.0** for tool-calling agents (deterministic, follows schema)
|
||||||
|
2. **Chain of Thought (CoT)** - force step-by-step reasoning
|
||||||
|
3. **Negative constraints** - tell model what NOT to do (Nemo responds better)
|
||||||
|
4. **Explicit tool descriptions** - verbose docstrings with "never estimate yourself"
|
||||||
|
5. **"Strictly tool-based assistant"** pattern - NO internal knowledge claim
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Experiment Log
|
||||||
|
|
||||||
|
### Baseline (v1.8.6)
|
||||||
|
- **Date**: 2025-12-17
|
||||||
|
- **Configuration**: Default temperature, improved prompt requiring list_devices first
|
||||||
|
- **Results**:
|
||||||
|
- Called list_devices first ✓
|
||||||
|
- Still hallucinated `light.study_desk` despite seeing list with only `light.study` and `light.study_main`
|
||||||
|
- Partial success: turned off `light.study_main`, failed on hallucinated entity
|
||||||
|
- **Success rate**: ~50% (1 of 2 study lights controlled correctly)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 1: Temperature 0.0
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**: Set `model_settings=ModelSettings(temperature=0.0)` for Housekeeper
|
||||||
|
- **Hypothesis**: Deterministic output will force model to use exact entity IDs from tool results
|
||||||
|
- **Results**:
|
||||||
|
|
||||||
|
**Study lights test:**
|
||||||
|
- Called `list_devices()` first ✓ (but no domain filter)
|
||||||
|
- Used wrong parameter `device_id` instead of `entity_id` (recovered after validation error)
|
||||||
|
- Only identified `light.studeerlamp` as "study" related (Dutch name)
|
||||||
|
- **Missed `light.study` and `light.study_main`** - didn't match English "study"
|
||||||
|
- Turned off 1 wrong light, missed 2 actual study lights
|
||||||
|
|
||||||
|
**Kitchen lights test:**
|
||||||
|
- Called `list_devices()` first ✓ (no domain filter)
|
||||||
|
- Saw full device list including `light.kitchen`
|
||||||
|
- Used wrong parameter `device_id` instead of `entity_id` (recovered after validation)
|
||||||
|
- After correction, dropped domain prefix: used `kitchen` instead of `light.kitchen`
|
||||||
|
- 404 error - device not found
|
||||||
|
|
||||||
|
- **Success rate**: 0% (no target lights successfully controlled)
|
||||||
|
- **Observations**:
|
||||||
|
- Temperature 0.0 alone is insufficient
|
||||||
|
- Model consistently confuses `device_id` vs `entity_id` parameter name
|
||||||
|
- After validation error correction, model truncates entity_id (drops domain prefix)
|
||||||
|
- Semantic matching of room names to devices is weak
|
||||||
|
- Model doesn't understand entity_id format: `domain.name`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 2: Negative Constraints + CoT
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**: Complete prompt rewrite with:
|
||||||
|
- "You have NO Internal Knowledge" - negative framing
|
||||||
|
- Explicit entity_id format with WRONG/RIGHT examples
|
||||||
|
- Step-by-step process (ALWAYS FOLLOW)
|
||||||
|
- Explicit parameter names section
|
||||||
|
- "What NOT To Do" negative constraints
|
||||||
|
- **Hypothesis**: Negative constraints work better with Mistral-Nemo
|
||||||
|
- **Results**:
|
||||||
|
|
||||||
|
**Study lights test:**
|
||||||
|
- Called `list_devices(domain="light")` ✓ with domain filter (improvement!)
|
||||||
|
- Still used `device_id` first, recovered to `entity_id` after validation error
|
||||||
|
- After recovery, used correct full format: `light.studeerlamp`
|
||||||
|
- **Still only matched `studeerlamp` not `light.study` or `light.study_main`**
|
||||||
|
|
||||||
|
**Kitchen lights test:**
|
||||||
|
- Called `list_devices(domain="light")` ✓
|
||||||
|
- Called `turn_off(entity_id="light.kitchen")` ✓ correct format!
|
||||||
|
- All 4 kitchen lights turned off (light.kitchen is a group)
|
||||||
|
- **100% success for kitchen!**
|
||||||
|
|
||||||
|
- **Success rate**:
|
||||||
|
- Study: 0% (wrong semantic match)
|
||||||
|
- Kitchen: 100% (4/4 lights off)
|
||||||
|
- Combined: ~50% (1 of 2 tests successful)
|
||||||
|
- **Observations**:
|
||||||
|
- Domain filter now consistently used ✓
|
||||||
|
- Entity_id format correct after recovery ✓
|
||||||
|
- Semantic matching still fails for "study" → prefers Dutch "studeerlamp" over English "study"
|
||||||
|
- Parameter name confusion persists (`device_id` vs `entity_id`)
|
||||||
|
- Simple room names (kitchen) work; mixed language fails (study/studeerlamp)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 3: Temperature 0.1 + Explicit Tool Docstrings
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**:
|
||||||
|
- Temperature 0.1
|
||||||
|
- Updated turn_on/turn_off docstrings with explicit `entity_id=` in examples
|
||||||
|
- **Results**:
|
||||||
|
- Still uses `device_id` first, recovers to `entity_id` after validation
|
||||||
|
- Still picks wrong entity (studeerlamp over study)
|
||||||
|
- **Success rate**: 0%
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 4: Room Group Priority (with explicit examples)
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**: Updated prompt with:
|
||||||
|
- Explicit instruction: "Look for EXACT match `light.<room_name>` first!"
|
||||||
|
- Concrete examples: "For 'study lights' → look for `light.study`"
|
||||||
|
- Working example showing `turn_off(entity_id="light.study")`
|
||||||
|
- **Hypothesis**: Explicit examples will guide model to use room groups
|
||||||
|
- **Results**:
|
||||||
|
|
||||||
|
**Test 1 & 2 (consecutive):**
|
||||||
|
- Called `list_devices(domain="light")` ✓
|
||||||
|
- Device list clearly shows `light.study` at the bottom
|
||||||
|
- First call: `turn_off({"devices":["studeerlamp"]})` - wrong param AND wrong device
|
||||||
|
- After validation error: `turn_off(entity_id="light.studeerlamp")` - correct param, still wrong device
|
||||||
|
- **Completely ignored `light.study` despite prompt explicitly saying to use it**
|
||||||
|
|
||||||
|
- **Success rate**: 0% (wrong device controlled)
|
||||||
|
- **Observations**:
|
||||||
|
- Model ignores explicit step-by-step instructions in favor of substring matching
|
||||||
|
- Dutch "studeerlamp" contains "studer" which the model prefers over exact "study" match
|
||||||
|
- Even when prompt has a literal example `turn_off(entity_id="light.study")`, model uses `light.studeerlamp`
|
||||||
|
- Positional bias possible - `light.study` appears at end of 21-item list
|
||||||
|
- **Fundamental limitation**: Mistral-Nemo cannot follow explicit matching rules
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 5: Room Groups First (Tool Output Ordering)
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**: Modified `list_devices` to sort room groups to top of list using HA attributes (`is_hue_group`, `hue_type="room"`)
|
||||||
|
- **Hypothesis**: Positional bias - model focuses on items earlier in list
|
||||||
|
- **Results**:
|
||||||
|
- Room groups (`light.study`, `light.kitchen`, etc.) now appear first in device list
|
||||||
|
- Combined with improved prompt, model now consistently uses room groups
|
||||||
|
- **70% success rate** (7/10 tests) with default q4 quantization
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 6: Model Quantization (q5_1)
|
||||||
|
- **Date**: 2025-12-18
|
||||||
|
- **Change**: Upgraded from default Mistral-Nemo quantization (q4) to `mistral-nemo:12b-instruct-2407-q5_1`
|
||||||
|
- **Hypothesis**: Higher precision weights improve tool calling accuracy
|
||||||
|
- **Results**:
|
||||||
|
|
||||||
|
| Test | Action | Result |
|
||||||
|
|------|--------|--------|
|
||||||
|
| 1 | Turn off study | PASS |
|
||||||
|
| 2 | Turn on study | PASS |
|
||||||
|
| 3 | Toggle study | PASS |
|
||||||
|
| 4 | Turn off kitchen | PASS |
|
||||||
|
| 5 | Turn on kitchen | PASS |
|
||||||
|
| 6 | Toggle kitchen | PASS |
|
||||||
|
| 7 | Turn off bedroom | PASS |
|
||||||
|
| 8 | Turn on bedroom | PASS |
|
||||||
|
| 9 | Turn off living room | PASS |
|
||||||
|
| 10 | Turn on living room | PASS |
|
||||||
|
|
||||||
|
- **Success rate**: **100%** (10/10 tests)
|
||||||
|
- **Observations**:
|
||||||
|
- q5_1 quantization dramatically improves tool calling accuracy
|
||||||
|
- All room groups correctly identified and used
|
||||||
|
- No parameter confusion (`entity_id` used correctly)
|
||||||
|
- No entity_id truncation issues
|
||||||
|
- Toggle operations now work reliably
|
||||||
|
- Model fits within 10GB VRAM (q6 did not)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Experiment 7: Device List in System Prompt (Context Injection)
|
||||||
|
- **Date**: [PENDING]
|
||||||
|
- **Change**: Store device list in database (per user/household) and inject into system prompt
|
||||||
|
- **Approach**:
|
||||||
|
1. Periodically sync device list from Home Assistant to PostgreSQL
|
||||||
|
2. On each Housekeeper invocation, fetch device list and include in prompt
|
||||||
|
3. Remove need for model to call list_devices() - just match from context
|
||||||
|
- **Hypothesis**:
|
||||||
|
- Eliminates tool call step where errors occur
|
||||||
|
- Reduces context size by not returning full device list as tool output
|
||||||
|
- Makes entity matching a language task (in prompt) rather than tool result parsing
|
||||||
|
- **Trade-offs**:
|
||||||
|
- Stale data if sync is infrequent
|
||||||
|
- Prompt size increase (but less than tool call response)
|
||||||
|
- Need sync mechanism and storage
|
||||||
|
- **Results**: [TO BE RECORDED]
|
||||||
|
- **Success rate**: [TO BE RECORDED]
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key Problem Identified (Solved)
|
||||||
|
|
||||||
|
The model struggled with:
|
||||||
|
1. **Parameter schema adherence** - uses `device_id` when schema requires `entity_id`
|
||||||
|
2. **Value preservation** - truncates values after validation errors (drops `light.` prefix)
|
||||||
|
3. **Semantic matching** - prefers substring matches ("studeerlamp" contains "studer") over exact matches (`light.study`)
|
||||||
|
4. **Following explicit instructions** - ignores step-by-step processes even when examples are provided
|
||||||
|
5. **Positional bias** - may not "see" items at the end of long lists
|
||||||
|
|
||||||
|
**Solution**: These issues were resolved by:
|
||||||
|
1. Using q5_1 quantization instead of default q4 (higher precision weights)
|
||||||
|
2. Sorting room groups to top of device list (address positional bias)
|
||||||
|
3. Explicit prompt guidance with negative constraints and examples
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Potential Next Experiments
|
||||||
|
|
||||||
|
### Experiment 5: Room Groups First (List Ordering)
|
||||||
|
- **Hypothesis**: Positional bias - model focuses on items earlier in list
|
||||||
|
- **Change**: Sort device list to put room groups (entities matching `light.<single_word>`) at the TOP
|
||||||
|
- **Effort**: Low - modify list_devices output formatting
|
||||||
|
- **Risk**: May affect other use cases where individual devices are needed
|
||||||
|
|
||||||
|
### Experiment 6: Simplified Device List Format
|
||||||
|
- **Hypothesis**: Markdown formatting adds noise that confuses the model
|
||||||
|
- **Change**: Return simple list: `light.study (Study - GROUP), light.study_main (Ceiling light), ...`
|
||||||
|
- **Effort**: Low - modify list_devices output
|
||||||
|
- **Risk**: Less human-readable responses
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Learnings to Apply Elsewhere
|
||||||
|
|
||||||
|
1. **Quantization matters** - q5_1 dramatically outperforms q4 for tool calling (100% vs 70%)
|
||||||
|
2. **Positional bias is real** - sort important items to top of lists
|
||||||
|
3. **Smaller models need simpler workflows** - fewer tool calls, more context injection
|
||||||
|
4. **Validation errors don't teach** - model often makes worse mistakes on retry
|
||||||
|
5. **Entity IDs are hard** - domain.name format confuses the model
|
||||||
|
6. **Consider pre-computation** - move matching logic to code, not LLM
|
||||||
|
7. **Use explicit negative constraints** - "NEVER do X" works better than "always do Y"
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- Librarian may need higher temperature for creative synthesis
|
||||||
|
- All "action" agents (Housekeeper, future agents) should use low temperature
|
||||||
|
- Consider testing with Gemma 2 9B for better function calling (Google, open weights)
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -10,7 +10,7 @@ This document establishes the foundational philosophy and architectural patterns
|
|||||||
- When new architectural insights require rethinking core principles
|
- When new architectural insights require rethinking core principles
|
||||||
|
|
||||||
**When NOT to modify this document**:
|
**When NOT to modify this document**:
|
||||||
- During implementation of these patterns (use README.md, AGENTS.md, or code comments for technical details)
|
- During implementation of these patterns (use README.md, CLAUDE.md, or code comments for technical details)
|
||||||
- For adding new household members or capabilities within the existing pattern
|
- For adding new household members or capabilities within the existing pattern
|
||||||
- For tactical decisions about specific technologies or tools
|
- For tactical decisions about specific technologies or tools
|
||||||
|
|
||||||
@@ -270,7 +270,7 @@ The user never directly interacts with the Steward or individual expert agents
|
|||||||
|
|
||||||
**Related Documents**:
|
**Related Documents**:
|
||||||
- **README.md**: User-facing documentation and usage guide
|
- **README.md**: User-facing documentation and usage guide
|
||||||
- **AGENTS.md**: LLM agent development guidelines and technical patterns
|
- **CLAUDE.md**: LLM agent development guidelines and technical patterns
|
||||||
- **CHANGELOG.md**: Version history and implemented features
|
- **CHANGELOG.md**: Version history and implemented features
|
||||||
|
|
||||||
---
|
---
|
||||||
@@ -0,0 +1,348 @@
|
|||||||
|
# Tatlock Integration Guide
|
||||||
|
|
||||||
|
Implementation instructions for integrating Library Desk search and content extraction endpoints into the Tatlock project.
|
||||||
|
|
||||||
|
## Base Configuration
|
||||||
|
|
||||||
|
```
|
||||||
|
BASE_URL: http://library-desk:8089 (or your deployment URL)
|
||||||
|
AUTH_HEADER: Authorization: Bearer <LIBRARY_API_KEY>
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. RAG Search Endpoint
|
||||||
|
|
||||||
|
**Use case:** Librarian needs to research a topic by searching the web.
|
||||||
|
|
||||||
|
### Endpoint
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /rag/search
|
||||||
|
```
|
||||||
|
|
||||||
|
### Request
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"query": "Python async programming best practices",
|
||||||
|
"search_type": "web",
|
||||||
|
"limit": 10,
|
||||||
|
"user": "tatlock-librarian"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
| Field | Type | Default | Description |
|
||||||
|
|-------|------|---------|-------------|
|
||||||
|
| `query` | string | required | Search query (1-500 chars) |
|
||||||
|
| `search_type` | enum | `"web"` | `"web"`, `"news"`, or `"images"` |
|
||||||
|
| `limit` | int | 10 | Results to return (1-20) |
|
||||||
|
| `user` | string | `"default"` | User identifier for tracking |
|
||||||
|
|
||||||
|
### Response
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"query": "Python async programming best practices",
|
||||||
|
"search_type": "web",
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"title": "Async IO in Python: A Complete Walkthrough",
|
||||||
|
"url": "https://realpython.com/async-io-python/",
|
||||||
|
"content": "Full extracted article text via Trafilatura (~2000 chars max)...",
|
||||||
|
"snippet": "Original search engine snippet (150-300 chars)...",
|
||||||
|
"source": "realpython.com",
|
||||||
|
"published_date": "2023-05-15"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"total_results": 10,
|
||||||
|
"search_time_ms": 2340,
|
||||||
|
"sources_summary": "## Sources\n- [Async IO in Python](https://realpython.com/async-io-python/)\n- ..."
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Key Fields for Tatlock
|
||||||
|
|
||||||
|
| Field | Usage |
|
||||||
|
|-------|-------|
|
||||||
|
| `results[].content` | Full extracted text - use this for LLM context |
|
||||||
|
| `results[].snippet` | Fallback if content extraction failed |
|
||||||
|
| `sources_summary` | Pre-formatted markdown for citations |
|
||||||
|
|
||||||
|
### Error Handling
|
||||||
|
|
||||||
|
| HTTP Code | Meaning | Action |
|
||||||
|
|-----------|---------|--------|
|
||||||
|
| 400 | Invalid query | Check query length/format |
|
||||||
|
| 502 | SearXNG unavailable | Retry with backoff |
|
||||||
|
| 504 | Search timeout | Retry or reduce limit |
|
||||||
|
| 500 | Internal error | Log and notify |
|
||||||
|
|
||||||
|
### Example Usage (Python)
|
||||||
|
|
||||||
|
```python
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
async def search_web(query: str, limit: int = 10) -> dict:
|
||||||
|
async with httpx.AsyncClient() as client:
|
||||||
|
response = await client.post(
|
||||||
|
f"{BASE_URL}/rag/search",
|
||||||
|
headers={"Authorization": f"Bearer {API_KEY}"},
|
||||||
|
json={
|
||||||
|
"query": query,
|
||||||
|
"search_type": "web",
|
||||||
|
"limit": limit,
|
||||||
|
"user": "tatlock-librarian"
|
||||||
|
},
|
||||||
|
timeout=30.0
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
return response.json()
|
||||||
|
|
||||||
|
# Usage
|
||||||
|
results = await search_web("machine learning transformers")
|
||||||
|
for r in results["results"]:
|
||||||
|
# Prefer full content, fall back to snippet
|
||||||
|
text = r["content"] or r["snippet"]
|
||||||
|
print(f"{r['title']}: {len(text)} chars")
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. Content Extraction Endpoint
|
||||||
|
|
||||||
|
**Use case:** Librarian has a specific URL and needs to read its content.
|
||||||
|
|
||||||
|
### Single URL Extraction
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /content/extract
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Request
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"url": "https://example.com/article",
|
||||||
|
"include_metadata": true,
|
||||||
|
"max_length": 2000
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Response
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"result": {
|
||||||
|
"url": "https://example.com/article",
|
||||||
|
"title": "Article Title",
|
||||||
|
"content": "Extracted main text content...",
|
||||||
|
"author": "John Doe",
|
||||||
|
"date": "2024-01-15",
|
||||||
|
"language": "en",
|
||||||
|
"success": true,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
"extraction_time_ms": 1250
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Batch URL Extraction
|
||||||
|
|
||||||
|
```
|
||||||
|
POST /content/extract/batch
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Request
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"urls": [
|
||||||
|
"https://example.com/article1",
|
||||||
|
"https://example.com/article2",
|
||||||
|
"https://example.com/article3"
|
||||||
|
],
|
||||||
|
"include_metadata": true,
|
||||||
|
"max_length": 2000
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Response
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"url": "https://example.com/article1",
|
||||||
|
"title": "Article 1",
|
||||||
|
"content": "Extracted content...",
|
||||||
|
"success": true,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"url": "https://example.com/article2",
|
||||||
|
"title": null,
|
||||||
|
"content": "",
|
||||||
|
"success": false,
|
||||||
|
"error": "Connection timeout"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"total_urls": 3,
|
||||||
|
"successful": 2,
|
||||||
|
"failed": 1,
|
||||||
|
"extraction_time_ms": 3500
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. Error Pattern: Soft Failures
|
||||||
|
|
||||||
|
> **Important:** Content extraction uses a **soft failure pattern** - individual URL failures do NOT throw HTTP errors.
|
||||||
|
|
||||||
|
### Why Soft Failures?
|
||||||
|
|
||||||
|
When extracting content from multiple URLs (batch) or even single URLs:
|
||||||
|
- Some sites block bots
|
||||||
|
- Some URLs are temporarily down
|
||||||
|
- Some pages have no extractable content
|
||||||
|
|
||||||
|
Instead of failing the entire request, we return:
|
||||||
|
- `success: true/false` per result
|
||||||
|
- `error: "reason"` when failed
|
||||||
|
- Empty `content: ""` on failure
|
||||||
|
|
||||||
|
### Handling Soft Failures
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def extract_with_fallback(url: str) -> str:
|
||||||
|
response = await client.post(
|
||||||
|
f"{BASE_URL}/content/extract",
|
||||||
|
headers={"Authorization": f"Bearer {API_KEY}"},
|
||||||
|
json={"url": url}
|
||||||
|
)
|
||||||
|
response.raise_for_status() # Only throws on 4xx/5xx
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
result = data["result"]
|
||||||
|
|
||||||
|
if result["success"]:
|
||||||
|
return result["content"]
|
||||||
|
else:
|
||||||
|
# Log the failure, return empty or handle gracefully
|
||||||
|
logger.warning(f"Extraction failed for {url}: {result['error']}")
|
||||||
|
return "" # Or raise, or use cached version, etc.
|
||||||
|
```
|
||||||
|
|
||||||
|
### Batch Processing Example
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def extract_batch_with_stats(urls: list[str]) -> dict:
|
||||||
|
response = await client.post(
|
||||||
|
f"{BASE_URL}/content/extract/batch",
|
||||||
|
headers={"Authorization": f"Bearer {API_KEY}"},
|
||||||
|
json={"urls": urls, "max_length": 3000}
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
|
||||||
|
# Separate successful and failed
|
||||||
|
successful = [r for r in data["results"] if r["success"]]
|
||||||
|
failed = [r for r in data["results"] if not r["success"]]
|
||||||
|
|
||||||
|
if failed:
|
||||||
|
logger.warning(f"{len(failed)} URLs failed extraction:")
|
||||||
|
for f in failed:
|
||||||
|
logger.warning(f" {f['url']}: {f['error']}")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"contents": {r["url"]: r["content"] for r in successful},
|
||||||
|
"failed_urls": [f["url"] for f in failed],
|
||||||
|
"success_rate": data["successful"] / data["total_urls"]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Recommended Patterns for Tatlock
|
||||||
|
|
||||||
|
### Research Flow
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def librarian_research(topic: str) -> dict:
|
||||||
|
"""
|
||||||
|
Full research flow: search + extract additional context.
|
||||||
|
"""
|
||||||
|
# 1. Search for relevant pages
|
||||||
|
search_results = await search_web(topic, limit=10)
|
||||||
|
|
||||||
|
# 2. RAG search already includes extracted content
|
||||||
|
# Only extract more if you need deeper content
|
||||||
|
|
||||||
|
# 3. Build context for LLM
|
||||||
|
context_parts = []
|
||||||
|
for r in search_results["results"]:
|
||||||
|
content = r["content"] or r["snippet"]
|
||||||
|
if content:
|
||||||
|
context_parts.append(f"## {r['title']}\nSource: {r['url']}\n\n{content}")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"context": "\n\n---\n\n".join(context_parts),
|
||||||
|
"sources": search_results["sources_summary"],
|
||||||
|
"result_count": search_results["total_results"]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Reading a Specific Page
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def librarian_read_page(url: str) -> str:
|
||||||
|
"""
|
||||||
|
Read a specific URL the user provided.
|
||||||
|
"""
|
||||||
|
response = await client.post(
|
||||||
|
f"{BASE_URL}/content/extract",
|
||||||
|
headers={"Authorization": f"Bearer {API_KEY}"},
|
||||||
|
json={"url": url, "max_length": 5000} # Longer for deep reads
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
result = response.json()["result"]
|
||||||
|
|
||||||
|
if not result["success"]:
|
||||||
|
raise ValueError(f"Could not read page: {result['error']}")
|
||||||
|
|
||||||
|
# Format for LLM
|
||||||
|
header = f"# {result['title'] or 'Untitled'}\n"
|
||||||
|
if result["author"]:
|
||||||
|
header += f"Author: {result['author']}\n"
|
||||||
|
if result["date"]:
|
||||||
|
header += f"Date: {result['date']}\n"
|
||||||
|
|
||||||
|
return header + "\n" + result["content"]
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Rate Limits & Best Practices
|
||||||
|
|
||||||
|
| Recommendation | Reason |
|
||||||
|
|----------------|--------|
|
||||||
|
| Use `limit: 5-10` for searches | More results = longer extraction time |
|
||||||
|
| Batch URLs when possible | More efficient than sequential calls |
|
||||||
|
| Max 20 URLs per batch | Server limit |
|
||||||
|
| Set reasonable timeouts (30s) | Content extraction can be slow |
|
||||||
|
| Cache results client-side | Same URL rarely changes content |
|
||||||
|
| Use `user` parameter | Helps with debugging and rate limiting |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 6. Quick Reference
|
||||||
|
|
||||||
|
| Endpoint | Method | Use Case |
|
||||||
|
|----------|--------|----------|
|
||||||
|
| `/rag/search` | POST | Search web + get extracted content |
|
||||||
|
| `/content/extract` | POST | Read a single URL |
|
||||||
|
| `/content/extract/batch` | POST | Read multiple URLs |
|
||||||
|
| `/health` | GET | Check service status |
|
||||||
+208
@@ -0,0 +1,208 @@
|
|||||||
|
# Tatlock Implementation Roadmap
|
||||||
|
|
||||||
|
> **Reference**: See [philosophy.md](philosophy.md) for the target architecture and vision
|
||||||
|
|
||||||
|
This document tracks open/planned work. Completed phases have been removed.
|
||||||
|
|
||||||
|
## Current State (v2.0.5)
|
||||||
|
|
||||||
|
**What we have**:
|
||||||
|
- OpenAI-compatible API (Responses API + Chat Completions)
|
||||||
|
- Two-tier architecture (Steward → Tatlock)
|
||||||
|
- Household staff: Tatlock (Butler), Steward, Librarian, Biographer
|
||||||
|
- Core tools: Calculator, Date/Time, Web search (SearXNG)
|
||||||
|
- Memory system: Qdrant (vector), Redis (session cache), multi-tenancy via ContextVar
|
||||||
|
- Dual backend: Ollama/gemma4 (primary) + Claude (fallback)
|
||||||
|
- 439 tests with good coverage
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 4: Expert Household Staff — Remaining Agents
|
||||||
|
|
||||||
|
**Goal**: Implement remaining domain-specific expert agents
|
||||||
|
|
||||||
|
### Planned Agents
|
||||||
|
|
||||||
|
1. **The Developer** (Software Development)
|
||||||
|
- Code generation assistance
|
||||||
|
- Debugging support
|
||||||
|
- Documentation generation
|
||||||
|
- Architecture guidance
|
||||||
|
|
||||||
|
2. **The Handyman** (System Maintenance)
|
||||||
|
- System status queries
|
||||||
|
- Log analysis
|
||||||
|
- Basic troubleshooting
|
||||||
|
- Infrastructure monitoring
|
||||||
|
|
||||||
|
3. **The Secretary** (Scheduling & Organization)
|
||||||
|
- Calendar integration
|
||||||
|
- Task management
|
||||||
|
- Reminder system
|
||||||
|
- Schedule conflict detection
|
||||||
|
|
||||||
|
4. **The Housekeeper** (Home Automation)
|
||||||
|
- Home Assistant integration
|
||||||
|
- Device control interface
|
||||||
|
- Status queries
|
||||||
|
- Automation triggers
|
||||||
|
|
||||||
|
### Each Agent Includes
|
||||||
|
- Specialized prompt and personality
|
||||||
|
- Domain-specific tools
|
||||||
|
- MCP integration points (where applicable)
|
||||||
|
- Integration with Butler orchestration
|
||||||
|
|
||||||
|
### Success Criteria
|
||||||
|
- [ ] Each agent implemented as separate module
|
||||||
|
- [ ] Agents callable via tool framework
|
||||||
|
- [ ] Can invoke specialized models (e.g., Codestral for Developer)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 5: Persistence Layer — Database & Multi-Tenancy
|
||||||
|
|
||||||
|
**Goal**: Add persistent storage and multi-user support
|
||||||
|
|
||||||
|
### Deliverables
|
||||||
|
|
||||||
|
1. **PostgreSQL Integration**
|
||||||
|
- Docker compose configuration
|
||||||
|
- Database schema with tenant isolation
|
||||||
|
- Alembic migrations
|
||||||
|
- SQLAlchemy models
|
||||||
|
|
||||||
|
2. **Multi-Tenant Architecture**
|
||||||
|
- Tenant identification middleware
|
||||||
|
- Tenant-scoped database sessions
|
||||||
|
- User authentication system
|
||||||
|
- Per-tenant data isolation
|
||||||
|
|
||||||
|
3. **Core Data Models**
|
||||||
|
- Users and tenants
|
||||||
|
- Conversations and messages (migrate from in-memory)
|
||||||
|
- Agent interactions log
|
||||||
|
- System configuration and preferences
|
||||||
|
|
||||||
|
### Success Criteria
|
||||||
|
- [ ] PostgreSQL container running
|
||||||
|
- [ ] Multiple users authenticate separately
|
||||||
|
- [ ] Each user sees only their own data
|
||||||
|
- [ ] Conversations persist across restarts
|
||||||
|
- [ ] Database migrations work correctly
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 7: MCP (Model Context Protocol) Integration
|
||||||
|
|
||||||
|
**Goal**: Enable rich tool integrations via MCP
|
||||||
|
|
||||||
|
See also [claude-integration.md](claude-integration.md) for MCP server implementation details.
|
||||||
|
|
||||||
|
### Deliverables
|
||||||
|
|
||||||
|
1. **MCP Server Framework**
|
||||||
|
- MCP server implementation
|
||||||
|
- Tool registration via MCP
|
||||||
|
- Schema validation
|
||||||
|
- Error handling
|
||||||
|
|
||||||
|
2. **MCP Client in Agents**
|
||||||
|
- PydanticAI MCP integration
|
||||||
|
- Tool discovery from MCP servers
|
||||||
|
- Dynamic tool loading
|
||||||
|
|
||||||
|
3. **Initial MCP Tools**
|
||||||
|
- File system operations
|
||||||
|
- Database queries
|
||||||
|
- API integrations
|
||||||
|
- System commands
|
||||||
|
|
||||||
|
### Success Criteria
|
||||||
|
- [ ] MCP server running
|
||||||
|
- [ ] Tools exposed via MCP protocol
|
||||||
|
- [ ] Agents can discover and use MCP tools
|
||||||
|
- [ ] New tools addable without code changes
|
||||||
|
- [ ] MCP tools visible in Steward recommendations
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 8: Advanced Memory & Context — Remaining Work
|
||||||
|
|
||||||
|
**Goal**: Implement sophisticated context management and personalization
|
||||||
|
|
||||||
|
### Open Deliverables
|
||||||
|
|
||||||
|
1. **Context Management**
|
||||||
|
- Smart context window trimming
|
||||||
|
- Conversation branching
|
||||||
|
- Topic tracking
|
||||||
|
|
||||||
|
2. **Personalization**
|
||||||
|
- User preference learning
|
||||||
|
- Interaction pattern analysis
|
||||||
|
- Adaptive responses
|
||||||
|
- Custom agent personalities per user
|
||||||
|
|
||||||
|
### Success Criteria
|
||||||
|
- [ ] Conversations automatically embedded to Qdrant
|
||||||
|
- [ ] Memory improves over time (learning from interactions)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 9: Extended Household Staff
|
||||||
|
|
||||||
|
**Goal**: Add specialized agents for additional domains
|
||||||
|
|
||||||
|
### Future Agents
|
||||||
|
- **The Accountant** — Expense tracking, budgets, financial reports
|
||||||
|
- **The Chef** — Meal planning, recipes, nutrition tracking
|
||||||
|
- Others as needs emerge
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 10: User Experience Refinement
|
||||||
|
|
||||||
|
**Goal**: Polish the interaction experience
|
||||||
|
|
||||||
|
- Personality tuning and consistency
|
||||||
|
- Better progress indicators
|
||||||
|
- Response time improvements
|
||||||
|
- Streaming smoothness
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 11: Production Hardening
|
||||||
|
|
||||||
|
**Goal**: Make the system production-ready for homelab deployment
|
||||||
|
|
||||||
|
- Complete docker-compose stack
|
||||||
|
- Health checks and monitoring
|
||||||
|
- Authentication hardening and rate limiting
|
||||||
|
- Installation and troubleshooting documentation
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Dependencies
|
||||||
|
|
||||||
|
```
|
||||||
|
Phase 4 (Remaining Agents)
|
||||||
|
↓
|
||||||
|
Phase 5 (Database/Multi-Tenancy) ← Can be deferred
|
||||||
|
↓
|
||||||
|
Phase 7 (MCP) → Phase 8 (Advanced Memory)
|
||||||
|
↓
|
||||||
|
Phase 9 (Extended Staff) → Phase 10 (UX) → Phase 11 (Production)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Can Be Deferred**: Phase 5 until you need persistence
|
||||||
|
**Parallel Opportunities**: Phases 7 and 8 can overlap; 9 and 10 ongoing
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Next Steps
|
||||||
|
|
||||||
|
1. Implement The Developer agent for code assistance
|
||||||
|
2. Add Home Assistant integration for The Housekeeper
|
||||||
|
3. Integrate scheduling service for The Secretary
|
||||||
|
4. MCP server for external Claude access
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
# Steward Routing & Thinking — Findings
|
||||||
|
|
||||||
|
**Outcome: no change shipped.** The Steward stays on `gemma4:e2b` with model
|
||||||
|
thinking left at its default (on). Every alternative was measured and every one
|
||||||
|
loses. This document exists so the experiment is not repeated on the same
|
||||||
|
premise.
|
||||||
|
|
||||||
|
Run 2026-08-08 with `scripts/benchmark_routing.py` and
|
||||||
|
`scripts/fixtures/routing_fixtures.py` (40 labelled queries, one repeat per
|
||||||
|
cell, temperature 0.3 as production sends).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## The premise was wrong
|
||||||
|
|
||||||
|
The experiment was designed around an observation that the Steward pays ~300
|
||||||
|
tokens per turn for reasoning that is generated and thrown away: it calls
|
||||||
|
`/api/generate`, gemma4 reasons by default, and **no `thinking` field comes back
|
||||||
|
in the response**. Disabling thinking therefore looked close to free.
|
||||||
|
|
||||||
|
It is not. The reasoning is not discarded — it is emitted inline in `response`,
|
||||||
|
and it is what produces a correct `DELEGATE:` line. Those tokens are the work,
|
||||||
|
not waste. Suppressing them costs 12.5 points of routing accuracy.
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
| config | exact | under | over | tokens | latency | resident | predicted | co-resident with nomic |
|
||||||
|
|---|---|---|---|---|---|---|---|---|
|
||||||
|
| **e2b, thinking** *(production)* | **97.5%** | 2.5% | 0% | 361 | 5179 ms | 1778 MB | 7.8 GiB | yes |
|
||||||
|
| e2b, `think: false` | 85.0% | 12.5% | 5.0% | 48 | 1435 ms | 1778 MB | 7.8 GiB | yes |
|
||||||
|
| e4b, thinking | 100% | 0% | 0% | 192 | 4726 ms | 3089 MB | 10.6 GiB | **no** |
|
||||||
|
| e4b, `think: false` | 97.5% | 2.5% | 0% | 52 | 2269 ms | 3089 MB | 10.6 GiB | **no** |
|
||||||
|
|
||||||
|
`think: true` was also measured and landed within one fixture of the default on
|
||||||
|
both models, so production's implicit thinking is the same thing as asking for
|
||||||
|
it explicitly. Format compliance was 100% in every cell — a `DELEGATE:` line is
|
||||||
|
always emitted.
|
||||||
|
|
||||||
|
With 40 fixtures and one repeat, each result is worth 2.5 points, so the
|
||||||
|
97.5-vs-100 gaps are single fixtures and inside the noise. The latency and token
|
||||||
|
medians (40 calls each) and the e2b `think: false` degradation (6 failures with a
|
||||||
|
consistent mechanism) are the parts worth trusting.
|
||||||
|
|
||||||
|
## Why each alternative loses
|
||||||
|
|
||||||
|
**`think: false` on e2b** — 85% exact, and the failures are not random. All three
|
||||||
|
multi-capability fixtures under-route, each missing a second capability. Without
|
||||||
|
reasoning the model names one capability and stops decomposing. It is not
|
||||||
|
degraded across the board; it specifically stops handling compound requests,
|
||||||
|
which is where a user would most notice the Butler quietly doing half the job.
|
||||||
|
|
||||||
|
**e4b, either setting** — disqualified by memory, not by quality. Ollama predicts
|
||||||
|
**10.6 GiB** for it at 16k context. Maximum available on this card is ~7.9 GiB
|
||||||
|
(10.4 free − 2.0 GPU overhead − 0.46 minimum), so e4b *always* exceeds the budget
|
||||||
|
and evicts every co-resident before loading. Observed directly: loading it threw
|
||||||
|
out both `gemma4:e2b` and `nomic-embed-text`. Losing nomic means Tatlock memory
|
||||||
|
and library-desk thrash on every embedding call. Note this is not caused by the
|
||||||
|
2 GiB reservation — without it, available would be ~9.7 GiB, still under 10.6.
|
||||||
|
|
||||||
|
**Lower `OLLAMA_CONTEXT_LENGTH`** — the obvious way to free headroom, and it does
|
||||||
|
not work. Dropping 16384 → 2048, an 8× reduction, moved the prediction only from
|
||||||
|
7.8 to 6.7 GiB. The prediction is dominated by weights and batch size, not KV
|
||||||
|
cache. It would also truncate the Librarian's retrieved passages and webber's
|
||||||
|
code context for a 14% saving that funds nothing.
|
||||||
|
|
||||||
|
**Per-request `num_ctx`** — worse. A single request with a different `num_ctx`
|
||||||
|
reloads the shared runner, which **drops the `keep_alive: -1` pin** (expiry fell
|
||||||
|
from year-2318 to a 2-hour default) and evicts nomic. Three services share this
|
||||||
|
Ollama, so mixed context sizes are a thrash generator, and it fails silently.
|
||||||
|
|
||||||
|
**`OLLAMA_NUM_PARALLEL > 1`** — never viable here. e2b already predicts 7.8 GiB
|
||||||
|
against ~7.9 available, so there is no room for a second slot at any context
|
||||||
|
length. It is also set to 1 deliberately, to avoid batch overflow panics.
|
||||||
|
|
||||||
|
## What the two axes actually control
|
||||||
|
|
||||||
|
They do not interact, which is the useful part:
|
||||||
|
|
||||||
|
- **Model choice** governs VRAM and co-residency. e2b 1778 MB, e4b 3089 MB.
|
||||||
|
- **Think setting** governs tokens, latency and routing quality — and costs
|
||||||
|
**nothing** in VRAM. Verified: e2b is resident at 1778 MB with `think` unset,
|
||||||
|
true and false alike, because the KV cache is allocated for the full context at
|
||||||
|
load time and `think` is a per-request generation parameter.
|
||||||
|
|
||||||
|
So the only real question is whether 313 tokens and 3.7 seconds are worth 12.5
|
||||||
|
points of compound-query routing. On a turn that is already three sequential
|
||||||
|
Ollama calls, they are.
|
||||||
|
|
||||||
|
## Prerequisite: the extraction fix
|
||||||
|
|
||||||
|
These numbers are only meaningful because `_extract_capabilities` was fixed first
|
||||||
|
(commit `a905363`). It previously substring-matched capability *domains* across
|
||||||
|
the Steward's entire response, so ordinary English in the `REASON:` line selected
|
||||||
|
agents — "description" contains the housekeeper domain "script", "acknowledge"
|
||||||
|
contains "knowledge" and "know".
|
||||||
|
|
||||||
|
That made **prose length a routing input**. Benchmarking against it would have
|
||||||
|
shown `think: false` improving routing purely because shorter output produces
|
||||||
|
fewer accidental substring hits — a thinking policy derived from a parsing
|
||||||
|
artefact. The `adversarial` fixture group is regression coverage for exactly this.
|
||||||
|
|
||||||
|
## If this is revisited
|
||||||
|
|
||||||
|
The constraint is the single 11 GB card, not the model. A second inference host
|
||||||
|
(*forge*) removes it entirely, and e4b's 100% routing becomes reachable without
|
||||||
|
evicting anything. Re-run then; on this card the answer is settled.
|
||||||
|
|
||||||
|
`scripts/benchmark_routing.py` takes `--models`, `--think` and `--repeats`, and
|
||||||
|
restores GPU residency on exit — including on SIGTERM, which the first version
|
||||||
|
did not.
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
# Testing Improvements for LLM Outputs
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
LLM outputs are non-deterministic. Tests checking for exact string matches fail when the LLM writes "thirty-seven" instead of "37".
|
||||||
|
|
||||||
|
## Proposed Solutions
|
||||||
|
|
||||||
|
### 1. LLM-as-Judge Pattern
|
||||||
|
|
||||||
|
Use a smaller/faster model to evaluate semantic correctness:
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def llm_judge(output: str, criteria: str) -> bool:
|
||||||
|
"""Use LLM to evaluate if output meets criteria."""
|
||||||
|
prompt = f"""
|
||||||
|
Evaluate if this output is correct:
|
||||||
|
Output: {output}
|
||||||
|
Criteria: {criteria}
|
||||||
|
Answer only YES or NO.
|
||||||
|
"""
|
||||||
|
result = await judge_model.run(prompt)
|
||||||
|
return "YES" in result.output.upper()
|
||||||
|
|
||||||
|
# Usage in test:
|
||||||
|
assert await llm_judge(
|
||||||
|
response,
|
||||||
|
"The answer correctly states that sqrt(144) + 25 = 37"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Fuzzy/Regex Matching
|
||||||
|
|
||||||
|
For numeric answers, accept multiple representations:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import re
|
||||||
|
|
||||||
|
def contains_number(text: str, number: int) -> bool:
|
||||||
|
"""Check if text contains number in any form."""
|
||||||
|
patterns = [
|
||||||
|
rf'\b{number}\b', # Digit form
|
||||||
|
number_to_words(number), # Word form
|
||||||
|
]
|
||||||
|
return any(re.search(p, text, re.I) for p in patterns)
|
||||||
|
|
||||||
|
# Usage:
|
||||||
|
assert contains_number(response, 37) # Matches "37" or "thirty-seven"
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3. DeepEval Framework
|
||||||
|
|
||||||
|
```python
|
||||||
|
from deepeval.metrics import AnswerRelevancyMetric
|
||||||
|
from deepeval.test_case import LLMTestCase
|
||||||
|
|
||||||
|
def test_calculation():
|
||||||
|
test_case = LLMTestCase(
|
||||||
|
input="What is sqrt(144) + 25?",
|
||||||
|
actual_output=response,
|
||||||
|
expected_output="37"
|
||||||
|
)
|
||||||
|
metric = AnswerRelevancyMetric(threshold=0.7)
|
||||||
|
assert metric.measure(test_case)
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4. pytest-evals Plugin
|
||||||
|
|
||||||
|
Minimal pytest plugin for LLM testing with metrics collection.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install pytest-evals
|
||||||
|
```
|
||||||
|
|
||||||
|
### 5. Multiple Runs with Threshold
|
||||||
|
|
||||||
|
Run flaky tests multiple times and require majority pass:
|
||||||
|
|
||||||
|
```python
|
||||||
|
@pytest.mark.flaky(reruns=3, reruns_delay=1)
|
||||||
|
def test_llm_response():
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
Or custom:
|
||||||
|
|
||||||
|
```python
|
||||||
|
@pytest.mark.parametrize("run", range(3))
|
||||||
|
def test_llm_response(run):
|
||||||
|
...
|
||||||
|
# Aggregate results across runs
|
||||||
|
```
|
||||||
|
|
||||||
|
## Resources
|
||||||
|
|
||||||
|
- [DeepEval](https://github.com/confident-ai/deepeval) - LLM evaluation framework
|
||||||
|
- [pytest-evals](https://github.com/AlmogBaku/pytest-evals) - pytest plugin for LLM evals
|
||||||
|
- [LLM Testing Guide 2025](https://www.confident-ai.com/blog/llm-testing-in-2024-top-methods-and-strategies)
|
||||||
|
- [Testing LLM Applications - Langfuse](https://langfuse.com/blog/2025-10-21-testing-llm-applications)
|
||||||
|
|
||||||
|
## Implementation Priority
|
||||||
|
|
||||||
|
1. Add fuzzy number matching helper (quick win)
|
||||||
|
2. Evaluate DeepEval for complex output testing
|
||||||
|
3. Consider LLM-as-judge for semantic correctness
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
# Decisions, Questions, Rejected
|
||||||
|
|
||||||
|
This directory holds structured planning records that pql parses
|
||||||
|
into pql.db. Each record is a `### [DQR]-N: Title` heading inside
|
||||||
|
a markdown file. Files live in three per-type subdirectories:
|
||||||
|
|
||||||
|
- `decisions/<domain>.md` — confirmed design decisions
|
||||||
|
- `questions/<domain>.md` — open questions that may resolve into
|
||||||
|
decisions or rejected proposals
|
||||||
|
- `rejected/<domain>.md` — rejected proposals (kept for the audit
|
||||||
|
trail)
|
||||||
|
|
||||||
|
The parser infers domain from the filename stem and record type
|
||||||
|
from the parent subdirectory.
|
||||||
|
|
||||||
|
D-records that propose implementation work link to `initiative`-type
|
||||||
|
tickets via `decision_ref`. Run `pql decisions show <id>
|
||||||
|
--with-tickets` to inspect implementation status.
|
||||||
|
|
||||||
|
## Recommended domains
|
||||||
|
|
||||||
|
Start with this canonical set; create files as records land in
|
||||||
|
each domain:
|
||||||
|
|
||||||
|
- **architecture** — structural commitments (storage, layering,
|
||||||
|
languages, libraries)
|
||||||
|
- **process** — team workflow (commits, branches, releases, reviews)
|
||||||
|
- **design** — user-facing surface (UX, UI, public APIs)
|
||||||
|
- **coding-conventions** — team-internal code shape (style, lint,
|
||||||
|
file layout)
|
||||||
|
- **testing** — quality strategy (coverage, layers, gates)
|
||||||
|
|
||||||
|
You might also want, project-permitting:
|
||||||
|
|
||||||
|
- `accessibility` — if you ship user-facing software
|
||||||
|
- `security` — if you handle user data or network surfaces
|
||||||
|
- `licensing` — if you release open-source or commercial
|
||||||
|
- `documentation` — if user-docs are non-trivial
|
||||||
|
- `deployment` — if shipping is non-trivial
|
||||||
|
- `performance` — if you have perf budgets / SLOs
|
||||||
|
|
||||||
|
<!-- pql:records (auto-generated; do not edit manually) -->
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- _(none)_
|
||||||
|
|
||||||
|
## Open questions
|
||||||
|
|
||||||
|
- _(none)_
|
||||||
|
|
||||||
|
## Rejected
|
||||||
|
|
||||||
|
- _(none)_
|
||||||
File diff suppressed because it is too large
Load Diff
+63
-3
@@ -4,17 +4,72 @@ build-backend = "setuptools.build_meta"
|
|||||||
|
|
||||||
[project]
|
[project]
|
||||||
name = "tatlock"
|
name = "tatlock"
|
||||||
version = "0.2.5"
|
version = "2.4.3"
|
||||||
description = "OpenAI-compatible API with Ollama backend"
|
description = "OpenAI-compatible API with Ollama backend"
|
||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
dependencies = []
|
dependencies = [
|
||||||
|
"fastapi>=0.123,<0.124",
|
||||||
|
"uvicorn[standard]>=0.38,<0.39",
|
||||||
|
"pydantic>=2.11,<2.13",
|
||||||
|
"pydantic-settings>=2.12,<2.13",
|
||||||
|
"pydantic-ai-slim[openai,anthropic]>=1.27,<1.28",
|
||||||
|
# pydantic-ai 1.27 imports the private opentelemetry._events module,
|
||||||
|
# removed in opentelemetry-api 1.44 — cap until pydantic-ai is bumped
|
||||||
|
"opentelemetry-api>=1.30,<1.44",
|
||||||
|
"anthropic>=0.77,<1.0",
|
||||||
|
"httpx>=0.28,<0.29",
|
||||||
|
"sse-starlette>=3.0,<3.1",
|
||||||
|
"python-dotenv>=1.2,<1.3",
|
||||||
|
"starlette>=0.45,<0.46",
|
||||||
|
"redis[hiredis]>=5.2,<6.0",
|
||||||
|
"qdrant-client>=1.12,<2.0",
|
||||||
|
"structlog>=24.1,<25.0",
|
||||||
|
]
|
||||||
|
|
||||||
|
[project.optional-dependencies]
|
||||||
|
dev = [
|
||||||
|
"pytest>=8.3,<8.4",
|
||||||
|
"pytest-asyncio>=0.25,<0.26",
|
||||||
|
"pytest-cov>=6.0,<6.1",
|
||||||
|
"pytest-mock>=3.14,<3.15",
|
||||||
|
"ruff>=0.8,<0.9",
|
||||||
|
"mypy>=1.14,<1.15",
|
||||||
|
"faker>=34.0,<35.0",
|
||||||
|
"coverage[toml]>=7.7,<7.8",
|
||||||
|
]
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
|
testpaths = ["tests"]
|
||||||
|
python_files = ["test_*.py"]
|
||||||
|
python_classes = ["Test*"]
|
||||||
|
python_functions = ["test_*"]
|
||||||
asyncio_mode = "auto"
|
asyncio_mode = "auto"
|
||||||
|
asyncio_default_fixture_loop_scope = "function"
|
||||||
|
cache_dir = ".cache/pytest"
|
||||||
|
markers = [
|
||||||
|
"unit: Unit tests",
|
||||||
|
"integration: Integration tests",
|
||||||
|
"slow: Slow running tests",
|
||||||
|
"contract: Wire-level contract tests against live service boundaries",
|
||||||
|
]
|
||||||
|
addopts = [
|
||||||
|
"--verbose",
|
||||||
|
"--strict-markers",
|
||||||
|
"--tb=short",
|
||||||
|
"--cov=src",
|
||||||
|
"--cov-report=term-missing",
|
||||||
|
"--cov-report=html:build/coverage/html",
|
||||||
|
"--cov-report=xml:build/coverage/coverage.xml",
|
||||||
|
"--cov-branch",
|
||||||
|
]
|
||||||
|
filterwarnings = [
|
||||||
|
"ignore::DeprecationWarning",
|
||||||
|
]
|
||||||
|
|
||||||
[tool.coverage.run]
|
[tool.coverage.run]
|
||||||
source = ["src"]
|
source = ["src"]
|
||||||
branch = true
|
branch = true
|
||||||
|
data_file = "build/coverage/.coverage"
|
||||||
omit = [
|
omit = [
|
||||||
"*/tests/*",
|
"*/tests/*",
|
||||||
"*/__pycache__/*",
|
"*/__pycache__/*",
|
||||||
@@ -37,11 +92,15 @@ exclude_lines = [
|
|||||||
]
|
]
|
||||||
|
|
||||||
[tool.coverage.html]
|
[tool.coverage.html]
|
||||||
directory = "htmlcov"
|
directory = "build/coverage/html"
|
||||||
|
|
||||||
|
[tool.coverage.xml]
|
||||||
|
output = "build/coverage/coverage.xml"
|
||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
line-length = 100
|
line-length = 100
|
||||||
target-version = "py312"
|
target-version = "py312"
|
||||||
|
cache-dir = ".cache/ruff"
|
||||||
|
|
||||||
[tool.ruff.lint]
|
[tool.ruff.lint]
|
||||||
select = [
|
select = [
|
||||||
@@ -64,6 +123,7 @@ ignore = [
|
|||||||
|
|
||||||
[tool.mypy]
|
[tool.mypy]
|
||||||
python_version = "3.12"
|
python_version = "3.12"
|
||||||
|
cache_dir = ".cache/mypy"
|
||||||
warn_return_any = true
|
warn_return_any = true
|
||||||
warn_unused_configs = true
|
warn_unused_configs = true
|
||||||
disallow_untyped_defs = true
|
disallow_untyped_defs = true
|
||||||
|
|||||||
-28
@@ -1,28 +0,0 @@
|
|||||||
[pytest]
|
|
||||||
testpaths = tests
|
|
||||||
python_files = test_*.py
|
|
||||||
python_classes = Test*
|
|
||||||
python_functions = test_*
|
|
||||||
asyncio_mode = auto
|
|
||||||
asyncio_default_fixture_loop_scope = function
|
|
||||||
|
|
||||||
# Markers
|
|
||||||
markers =
|
|
||||||
unit: Unit tests
|
|
||||||
integration: Integration tests
|
|
||||||
slow: Slow running tests
|
|
||||||
|
|
||||||
# Coverage options (overridden by pyproject.toml)
|
|
||||||
addopts =
|
|
||||||
--verbose
|
|
||||||
--strict-markers
|
|
||||||
--tb=short
|
|
||||||
--cov=src
|
|
||||||
--cov-report=term-missing
|
|
||||||
--cov-report=html
|
|
||||||
--cov-report=xml
|
|
||||||
--cov-branch
|
|
||||||
|
|
||||||
# Ignore warnings from dependencies
|
|
||||||
filterwarnings =
|
|
||||||
ignore::DeprecationWarning
|
|
||||||
@@ -1,25 +0,0 @@
|
|||||||
# Development and Testing Dependencies
|
|
||||||
# Install with: pip install -r requirements.txt -r requirements-dev.txt
|
|
||||||
|
|
||||||
# Testing Framework
|
|
||||||
# Latest pytest with async support
|
|
||||||
pytest>=8.3,<8.4
|
|
||||||
pytest-asyncio>=0.25,<0.26
|
|
||||||
pytest-cov>=6.0,<6.1
|
|
||||||
|
|
||||||
# Test client for FastAPI
|
|
||||||
httpx>=0.28,<0.29 # Already in requirements.txt but needed for test client
|
|
||||||
|
|
||||||
# Code Quality
|
|
||||||
# Linting and formatting
|
|
||||||
ruff>=0.8,<0.9
|
|
||||||
|
|
||||||
# Type checking
|
|
||||||
mypy>=1.14,<1.15
|
|
||||||
|
|
||||||
# Testing utilities
|
|
||||||
pytest-mock>=3.14,<3.15
|
|
||||||
faker>=34.0,<35.0
|
|
||||||
|
|
||||||
# Coverage reporting
|
|
||||||
coverage[toml]>=7.7,<7.8
|
|
||||||
@@ -1,51 +0,0 @@
|
|||||||
# Core FastAPI framework and server
|
|
||||||
# FastAPI: Modern, fast web framework for building APIs
|
|
||||||
# Latest: 0.123.9 (Dec 4, 2025) - No known CVEs
|
|
||||||
fastapi>=0.123,<0.124
|
|
||||||
|
|
||||||
# ASGI server for running FastAPI
|
|
||||||
# Latest: 0.38.0 (Oct 18, 2025) - No known CVEs
|
|
||||||
# Note: Old versions had CVE-2020-7694/7695, but 0.38.0 is secure
|
|
||||||
uvicorn[standard]>=0.38,<0.39
|
|
||||||
|
|
||||||
# Additional dependencies
|
|
||||||
# Pydantic for data validation (comes with pydantic-ai but pinning explicitly)
|
|
||||||
# Updated to >=2.11 due to ag-ui-protocol dependency requirement
|
|
||||||
# Latest: 2.12.4 (Nov 5, 2025) - No known CVEs
|
|
||||||
pydantic>=2.11,<2.13
|
|
||||||
|
|
||||||
# AI/LLM integration
|
|
||||||
# PydanticAI: Agent framework for using Pydantic with LLMs
|
|
||||||
# Latest: 1.27.0 (Dec 5, 2025) - No known CVEs
|
|
||||||
# Supports Ollama backend out of the box
|
|
||||||
pydantic-ai>=1.27,<1.28
|
|
||||||
|
|
||||||
# HTTP client for Ollama communication
|
|
||||||
# Latest: 0.28.1 - No known CVEs
|
|
||||||
httpx>=0.28,<0.29
|
|
||||||
|
|
||||||
# Server-Sent Events for streaming responses
|
|
||||||
# Required for OpenAI-compatible streaming endpoints
|
|
||||||
# Latest: 3.0.2 (Oct 30, 2025) - No known CVEs
|
|
||||||
sse-starlette>=3.0,<3.1
|
|
||||||
|
|
||||||
# Configuration management
|
|
||||||
# Latest: 1.2.1 (Oct 26, 2025) - No known CVEs
|
|
||||||
python-dotenv>=1.2,<1.3
|
|
||||||
|
|
||||||
# ASGI toolkit (dependency of FastAPI, pinning for security)
|
|
||||||
starlette>=0.45,<0.46
|
|
||||||
|
|
||||||
# Redis for performance benchmarking and caching
|
|
||||||
# Latest: 5.2.1 (Dec 5, 2025) - No known CVEs
|
|
||||||
# hiredis: C parser for better performance
|
|
||||||
redis[hiredis]>=5.2,<6.0
|
|
||||||
|
|
||||||
# Structured logging for observability
|
|
||||||
# Latest: 24.4.0 (Aug 22, 2024) - No known CVEs
|
|
||||||
structlog>=24.1,<25.0
|
|
||||||
|
|
||||||
# Note on version locking strategy:
|
|
||||||
# Using >=X.Y,<X.(Y+1) format to lock to minor versions
|
|
||||||
# This protects against supply chain attacks while allowing patch updates
|
|
||||||
# Update regularly and review changelogs before upgrading minor versions
|
|
||||||
@@ -0,0 +1,235 @@
|
|||||||
|
"""
|
||||||
|
Benchmark Steward routing quality against model and thinking settings.
|
||||||
|
|
||||||
|
Talks to Ollama directly. No Tatlock server, no agents, no tools, nothing is
|
||||||
|
executed — the mutating fixtures ("turn on the lights", "update the wiki") only
|
||||||
|
ever produce a routing decision. That makes this cheap and repeatable, and it
|
||||||
|
isolates the question: does the Steward still pick the right capabilities when
|
||||||
|
the model reasons less?
|
||||||
|
|
||||||
|
The request body is byte-identical to StewardAgent._call_ollama, plus the
|
||||||
|
`think` flag under test, so a cell labelled `unset` is exactly what production
|
||||||
|
sends today.
|
||||||
|
|
||||||
|
Three thinking settings, because "on vs off" hides the interesting case:
|
||||||
|
|
||||||
|
unset what production sends now. gemma4 reasons by default, and the
|
||||||
|
response carries no `thinking` field, so those tokens are generated
|
||||||
|
and discarded.
|
||||||
|
true reasoning requested explicitly and returned in `thinking`.
|
||||||
|
false reasoning suppressed.
|
||||||
|
|
||||||
|
Scoring is deliberately asymmetric. A missing capability under-routes and the
|
||||||
|
Butler answers without a tool it needed; a spurious one over-routes, and that is
|
||||||
|
a real agent call — a stray librarian is a multi-second web search on a query
|
||||||
|
that asked for arithmetic. Over-routing is the predicted failure when thinking
|
||||||
|
is off, so `forbid` violations are reported separately rather than folded into
|
||||||
|
one accuracy number.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
.venv/bin/python scripts/benchmark_routing.py
|
||||||
|
.venv/bin/python scripts/benchmark_routing.py --models gemma4:e2b
|
||||||
|
.venv/bin/python scripts/benchmark_routing.py --think false --repeats 3
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import statistics
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
sys.path.insert(0, str(PROJECT_ROOT))
|
||||||
|
|
||||||
|
from scripts.fixtures.routing_fixtures import FIXTURES # noqa: E402
|
||||||
|
from scripts.ollama_residency import ( # noqa: E402
|
||||||
|
install_sigterm_handler,
|
||||||
|
residency_guard,
|
||||||
|
)
|
||||||
|
from src.agents.steward.agent import build_steward_prompt # noqa: E402
|
||||||
|
from src.agents.steward.service import _DELEGATE_LINE_RE, _extract_capabilities # noqa: E402
|
||||||
|
from src.core.startup import register_household_members # noqa: E402
|
||||||
|
|
||||||
|
OLLAMA_URL = "http://localhost:11434"
|
||||||
|
DEFAULT_MODELS = ["gemma4:e2b", "gemma4:e4b"]
|
||||||
|
DEFAULT_THINK = ["unset", "true", "false"]
|
||||||
|
RESULTS_DIR = PROJECT_ROOT / "logs"
|
||||||
|
|
||||||
|
|
||||||
|
def build_body(model: str, prompt: str, think: str) -> dict[str, Any]:
|
||||||
|
"""Mirror StewardAgent._call_ollama exactly, then add the flag under test."""
|
||||||
|
body: dict[str, Any] = {
|
||||||
|
"model": model,
|
||||||
|
"prompt": prompt,
|
||||||
|
"stream": False,
|
||||||
|
"options": {
|
||||||
|
"temperature": 0.3, # Lower = more consistent
|
||||||
|
"top_p": 0.9,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
if think != "unset":
|
||||||
|
body["think"] = think == "true"
|
||||||
|
return body
|
||||||
|
|
||||||
|
|
||||||
|
def call(client: httpx.Client, body: dict[str, Any]) -> dict[str, Any] | None:
|
||||||
|
try:
|
||||||
|
response = client.post(f"{OLLAMA_URL}/api/generate", json=body)
|
||||||
|
response.raise_for_status()
|
||||||
|
return response.json()
|
||||||
|
except Exception as exc: # noqa: BLE001 - a failed cell must not abort the run
|
||||||
|
print(f" ! {exc}", file=sys.stderr)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def score(fixture: dict, found: list[str]) -> dict[str, Any]:
|
||||||
|
expected = set(fixture["expect"])
|
||||||
|
forbidden = set(fixture["forbid"])
|
||||||
|
got = set(found)
|
||||||
|
missing = sorted(expected - got)
|
||||||
|
spurious = sorted(got & forbidden)
|
||||||
|
return {
|
||||||
|
"found": found,
|
||||||
|
"missing": missing,
|
||||||
|
"spurious": spurious,
|
||||||
|
# Exact only when everything expected arrived and nothing forbidden did.
|
||||||
|
"exact": not missing and not spurious,
|
||||||
|
"under_routed": bool(missing),
|
||||||
|
"over_routed": bool(spurious),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def run_cell(client: httpx.Client, model: str, think: str, repeats: int) -> list[dict[str, Any]]:
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
for fixture in FIXTURES:
|
||||||
|
prompt = build_steward_prompt(fixture["query"], [])
|
||||||
|
body = build_body(model, prompt, think)
|
||||||
|
for rep in range(repeats):
|
||||||
|
started = time.perf_counter()
|
||||||
|
data = call(client, body)
|
||||||
|
elapsed_ms = (time.perf_counter() - started) * 1000
|
||||||
|
if data is None:
|
||||||
|
rows.append({
|
||||||
|
"id": fixture["id"], "group": fixture["group"], "rep": rep,
|
||||||
|
"error": True, "exact": False, "under_routed": False, "over_routed": False,
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
|
||||||
|
text = data.get("response", "") or ""
|
||||||
|
found = _extract_capabilities(text)
|
||||||
|
rows.append({
|
||||||
|
"id": fixture["id"],
|
||||||
|
"group": fixture["group"],
|
||||||
|
"rep": rep,
|
||||||
|
"error": False,
|
||||||
|
"latency_ms": round(elapsed_ms, 1),
|
||||||
|
"eval_tokens": data.get("eval_count"),
|
||||||
|
"prompt_tokens": data.get("prompt_eval_count"),
|
||||||
|
# Did the model obey the documented output shape at all?
|
||||||
|
"has_delegate_line": bool(_DELEGATE_LINE_RE.search(text)),
|
||||||
|
# Whether reasoning came back, as opposed to being generated and dropped.
|
||||||
|
"thinking_returned": bool(data.get("thinking")),
|
||||||
|
"response_chars": len(text),
|
||||||
|
**score(fixture, found),
|
||||||
|
})
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def summarise(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||||
|
ok = [r for r in rows if not r["error"]]
|
||||||
|
if not ok:
|
||||||
|
return {"n": 0, "errors": len(rows)}
|
||||||
|
latencies = [r["latency_ms"] for r in ok]
|
||||||
|
tokens = [r["eval_tokens"] for r in ok if r["eval_tokens"] is not None]
|
||||||
|
return {
|
||||||
|
"n": len(ok),
|
||||||
|
"errors": len(rows) - len(ok),
|
||||||
|
"exact_pct": round(100 * sum(r["exact"] for r in ok) / len(ok), 1),
|
||||||
|
"under_routed_pct": round(100 * sum(r["under_routed"] for r in ok) / len(ok), 1),
|
||||||
|
"over_routed_pct": round(100 * sum(r["over_routed"] for r in ok) / len(ok), 1),
|
||||||
|
"format_ok_pct": round(100 * sum(r["has_delegate_line"] for r in ok) / len(ok), 1),
|
||||||
|
"thinking_returned_pct": round(100 * sum(r["thinking_returned"] for r in ok) / len(ok), 1),
|
||||||
|
"latency_ms_median": round(statistics.median(latencies), 1),
|
||||||
|
"latency_ms_mean": round(statistics.fmean(latencies), 1),
|
||||||
|
"eval_tokens_median": round(statistics.median(tokens), 1) if tokens else None,
|
||||||
|
"eval_tokens_total": sum(tokens) if tokens else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
install_sigterm_handler()
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--models", default=",".join(DEFAULT_MODELS))
|
||||||
|
parser.add_argument("--think", default=",".join(DEFAULT_THINK),
|
||||||
|
help="comma-separated subset of unset,true,false")
|
||||||
|
parser.add_argument("--repeats", type=int, default=1)
|
||||||
|
parser.add_argument("--timeout", type=float, default=180.0)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
models = [m.strip() for m in args.models.split(",") if m.strip()]
|
||||||
|
think_modes = [t.strip() for t in args.think.split(",") if t.strip()]
|
||||||
|
|
||||||
|
# build_steward_prompt reads the registry, and the registry is populated at
|
||||||
|
# application startup. Without this the prompt lists no capabilities and every
|
||||||
|
# cell scores zero for reasons that have nothing to do with the model.
|
||||||
|
register_household_members()
|
||||||
|
|
||||||
|
print(f"{len(FIXTURES)} fixtures x {len(models)} models x {len(think_modes)} think "
|
||||||
|
f"x {args.repeats} repeats = {len(FIXTURES) * len(models) * len(think_modes) * args.repeats} calls\n")
|
||||||
|
|
||||||
|
cells: dict[str, Any] = {}
|
||||||
|
# The guard restores production's pinned models however this exits — a
|
||||||
|
# finished run, a failed cell, Ctrl-C or SIGTERM.
|
||||||
|
with residency_guard(models_used=models), httpx.Client(timeout=args.timeout) as client:
|
||||||
|
for model in models:
|
||||||
|
# Absorb the cold load (~36s) outside the measurements.
|
||||||
|
print(f"warming {model} ...", flush=True)
|
||||||
|
call(client, build_body(model, "hi", "false"))
|
||||||
|
for think in think_modes:
|
||||||
|
key = f"{model}|think={think}"
|
||||||
|
print(f" {key} ...", end=" ", flush=True)
|
||||||
|
started = time.perf_counter()
|
||||||
|
rows = run_cell(client, model, think, args.repeats)
|
||||||
|
summary = summarise(rows)
|
||||||
|
cells[key] = {"summary": summary, "rows": rows}
|
||||||
|
print(f"exact={summary.get('exact_pct')}% "
|
||||||
|
f"over={summary.get('over_routed_pct')}% "
|
||||||
|
f"median={summary.get('latency_ms_median')}ms "
|
||||||
|
f"({time.perf_counter() - started:.0f}s)")
|
||||||
|
|
||||||
|
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ")
|
||||||
|
out = RESULTS_DIR / f"routing-bench-{stamp}.json"
|
||||||
|
out.write_text(json.dumps({
|
||||||
|
"generated_at": datetime.now(UTC).isoformat(),
|
||||||
|
"fixtures": len(FIXTURES),
|
||||||
|
"repeats": args.repeats,
|
||||||
|
"cells": cells,
|
||||||
|
}, indent=2))
|
||||||
|
|
||||||
|
print(f"\n{'cell':28} {'exact':>7} {'under':>7} {'over':>7} {'fmt':>6} {'tok':>7} {'ms':>8}")
|
||||||
|
print("-" * 76)
|
||||||
|
for key, cell in cells.items():
|
||||||
|
s = cell["summary"]
|
||||||
|
print(f"{key:28} {s.get('exact_pct'):>6}% {s.get('under_routed_pct'):>6}% "
|
||||||
|
f"{s.get('over_routed_pct'):>6}% {s.get('format_ok_pct'):>5}% "
|
||||||
|
f"{str(s.get('eval_tokens_median')):>7} {s.get('latency_ms_median'):>8}")
|
||||||
|
print(f"\nwritten to {out}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
sys.exit(main())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
# The residency guard has already run by the time this is caught;
|
||||||
|
# a traceback here would just bury its output.
|
||||||
|
print("\ninterrupted", file=sys.stderr)
|
||||||
|
sys.exit(130)
|
||||||
@@ -268,7 +268,7 @@ async def run_benchmarks(iterations: int = 10, verbose: bool = False):
|
|||||||
print(f" Max: {overall_max:.3f}s (target: ≤5.0s)")
|
print(f" Max: {overall_max:.3f}s (target: ≤5.0s)")
|
||||||
print(f" Avg: {overall_avg:.3f}s (target: ≤1.67s)")
|
print(f" Avg: {overall_avg:.3f}s (target: ≤1.67s)")
|
||||||
print(f"\n Recommendations:")
|
print(f"\n Recommendations:")
|
||||||
print(f" - Switch to a faster model (current: mistral-nemo)")
|
print(f" - Switch to a faster model (current: gemma4:e2b)")
|
||||||
print(f" - Reduce system prompt complexity")
|
print(f" - Reduce system prompt complexity")
|
||||||
print(f" - Limit tool calls (currently limited to 3)")
|
print(f" - Limit tool calls (currently limited to 3)")
|
||||||
print(f" - Consider caching household registry responses")
|
print(f" - Consider caching household registry responses")
|
||||||
|
|||||||
@@ -0,0 +1,559 @@
|
|||||||
|
"""
|
||||||
|
Benchmark tool calling across different Ollama models via Tatlock API.
|
||||||
|
|
||||||
|
Sends test prompts through the full Tatlock pipeline (Steward -> Orchestration
|
||||||
|
-> Synthesis) and records tool selection accuracy, latency, and response quality.
|
||||||
|
|
||||||
|
Between models, swaps OLLAMA_DEFAULT_MODEL in .env and waits for uvicorn
|
||||||
|
auto-reload. Requires the server to be running via ./wakeup.sh.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
.venv/bin/python scripts/benchmark_tool_calling.py
|
||||||
|
.venv/bin/python scripts/benchmark_tool_calling.py --models "gemma4:e4b,gemma4:e2b"
|
||||||
|
.venv/bin/python scripts/benchmark_tool_calling.py --iterations 3
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import statistics
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||||
|
from scripts.ollama_residency import install_sigterm_handler, residency_guard
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Configuration
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
API_BASE = "http://localhost:8777"
|
||||||
|
CHAT_URL = f"{API_BASE}/v1/chat/completions"
|
||||||
|
HEALTH_URL = f"{API_BASE}/health"
|
||||||
|
OLLAMA_URL = "http://localhost:11434"
|
||||||
|
ENV_PATH = Path(__file__).parent.parent / ".env"
|
||||||
|
|
||||||
|
DEFAULT_MODELS = ["mistral-nemo-large:latest", "gemma4:e4b", "gemma4:e2b"]
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Test scenarios
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Scenario:
|
||||||
|
name: str
|
||||||
|
prompt: str
|
||||||
|
expected_tool: str | None # None = no tool expected
|
||||||
|
# Patterns to check in the response text for indirect tool-use evidence
|
||||||
|
success_patterns: list[str] = field(default_factory=list)
|
||||||
|
category: str = "basic"
|
||||||
|
|
||||||
|
|
||||||
|
SCENARIOS = [
|
||||||
|
# --- Should call calculate_math ---
|
||||||
|
Scenario(
|
||||||
|
name="Simple arithmetic",
|
||||||
|
prompt="What is 144 divided by 12?",
|
||||||
|
expected_tool="calculate_math",
|
||||||
|
success_patterns=["12"],
|
||||||
|
category="calculator",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Square root",
|
||||||
|
prompt="What's the square root of 256?",
|
||||||
|
expected_tool="calculate_math",
|
||||||
|
success_patterns=["16"],
|
||||||
|
category="calculator",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Complex math",
|
||||||
|
prompt="Calculate pi times the square of 5",
|
||||||
|
expected_tool="calculate_math",
|
||||||
|
success_patterns=["78.5"], # pi * 25 ≈ 78.54
|
||||||
|
category="calculator",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Word problem",
|
||||||
|
prompt="If I have 3 bags with 17 apples each and I eat 4, how many apples do I have?",
|
||||||
|
expected_tool="calculate_math",
|
||||||
|
success_patterns=["47"],
|
||||||
|
category="calculator",
|
||||||
|
),
|
||||||
|
|
||||||
|
# --- Should call get_current_time ---
|
||||||
|
Scenario(
|
||||||
|
name="Current date",
|
||||||
|
prompt="What's today's date?",
|
||||||
|
expected_tool="get_current_time",
|
||||||
|
success_patterns=["2026"], # Should contain current year
|
||||||
|
category="datetime",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Current time",
|
||||||
|
prompt="What time is it right now?",
|
||||||
|
expected_tool="get_current_time",
|
||||||
|
success_patterns=[":"], # Time format contains colons
|
||||||
|
category="datetime",
|
||||||
|
),
|
||||||
|
|
||||||
|
# --- Should call calculate_date_offset ---
|
||||||
|
Scenario(
|
||||||
|
name="Relative date past",
|
||||||
|
prompt="What was the date 2 weeks ago?",
|
||||||
|
expected_tool="calculate_date_offset",
|
||||||
|
success_patterns=["2026"],
|
||||||
|
category="datetime",
|
||||||
|
),
|
||||||
|
|
||||||
|
# --- Should call calculate_time_difference ---
|
||||||
|
Scenario(
|
||||||
|
name="Date difference",
|
||||||
|
prompt="How many days between January 1st 2025 and March 15th 2025?",
|
||||||
|
expected_tool="calculate_time_difference",
|
||||||
|
success_patterns=["73", "74"], # 73 or 74 days
|
||||||
|
category="datetime",
|
||||||
|
),
|
||||||
|
|
||||||
|
# --- Should NOT call any tool ---
|
||||||
|
Scenario(
|
||||||
|
name="Greeting",
|
||||||
|
prompt="Hello! How are you?",
|
||||||
|
expected_tool=None,
|
||||||
|
success_patterns=["sir"], # Butler personality
|
||||||
|
category="no_tool",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Knowledge question",
|
||||||
|
prompt="What is the capital of France?",
|
||||||
|
expected_tool=None,
|
||||||
|
success_patterns=["Paris"],
|
||||||
|
category="no_tool",
|
||||||
|
),
|
||||||
|
Scenario(
|
||||||
|
name="Opinion request",
|
||||||
|
prompt="What do you think about rainy days?",
|
||||||
|
expected_tool=None,
|
||||||
|
category="no_tool",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Result tracking
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class RunResult:
|
||||||
|
scenario: str
|
||||||
|
model: str
|
||||||
|
iteration: int
|
||||||
|
latency: float
|
||||||
|
response_text: str
|
||||||
|
has_correct_answer: bool
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ModelStats:
|
||||||
|
model: str
|
||||||
|
results: list[RunResult] = field(default_factory=list)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def total(self) -> int:
|
||||||
|
return len(self.results)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def errors(self) -> int:
|
||||||
|
return sum(1 for r in self.results if r.error)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def accuracy(self) -> float:
|
||||||
|
valid = [r for r in self.results if not r.error]
|
||||||
|
if not valid:
|
||||||
|
return 0
|
||||||
|
return sum(1 for r in valid if r.has_correct_answer) / len(valid) * 100
|
||||||
|
|
||||||
|
@property
|
||||||
|
def avg_latency(self) -> float:
|
||||||
|
lats = [r.latency for r in self.results if not r.error]
|
||||||
|
return statistics.mean(lats) if lats else 0
|
||||||
|
|
||||||
|
@property
|
||||||
|
def p95_latency(self) -> float:
|
||||||
|
lats = sorted(r.latency for r in self.results if not r.error)
|
||||||
|
if not lats:
|
||||||
|
return 0
|
||||||
|
return lats[min(int(len(lats) * 0.95), len(lats) - 1)]
|
||||||
|
|
||||||
|
@property
|
||||||
|
def max_latency(self) -> float:
|
||||||
|
lats = [r.latency for r in self.results if not r.error]
|
||||||
|
return max(lats) if lats else 0
|
||||||
|
|
||||||
|
def category_accuracy(self, category: str) -> float:
|
||||||
|
cat_scenarios = {s.name for s in SCENARIOS if s.category == category}
|
||||||
|
valid = [r for r in self.results if not r.error and r.scenario in cat_scenarios]
|
||||||
|
if not valid:
|
||||||
|
return 0
|
||||||
|
return sum(1 for r in valid if r.has_correct_answer) / len(valid) * 100
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# .env manipulation
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def swap_model_in_env(model_name: str):
|
||||||
|
"""Swap OLLAMA_DEFAULT_MODEL in .env file."""
|
||||||
|
content = ENV_PATH.read_text()
|
||||||
|
content = re.sub(
|
||||||
|
r'^OLLAMA_DEFAULT_MODEL=.*$',
|
||||||
|
f'OLLAMA_DEFAULT_MODEL={model_name}',
|
||||||
|
content,
|
||||||
|
flags=re.MULTILINE,
|
||||||
|
)
|
||||||
|
ENV_PATH.write_text(content)
|
||||||
|
print(f" .env updated: OLLAMA_DEFAULT_MODEL={model_name}")
|
||||||
|
|
||||||
|
|
||||||
|
async def wait_for_server_reload(client: httpx.AsyncClient, timeout: float = 30):
|
||||||
|
"""Wait for uvicorn to auto-reload after .env change."""
|
||||||
|
# Give uvicorn a moment to detect the file change
|
||||||
|
await asyncio.sleep(3)
|
||||||
|
|
||||||
|
# Poll health endpoint
|
||||||
|
deadline = time.monotonic() + timeout
|
||||||
|
while time.monotonic() < deadline:
|
||||||
|
try:
|
||||||
|
r = await client.get(HEALTH_URL, timeout=5)
|
||||||
|
if r.status_code == 200:
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
|
||||||
|
raise TimeoutError("Server did not come back after reload")
|
||||||
|
|
||||||
|
|
||||||
|
async def warm_up_ollama_model(client: httpx.AsyncClient, model_name: str):
|
||||||
|
"""Send a throwaway request to load the model into VRAM."""
|
||||||
|
print(f" Warming up {model_name} in Ollama...", end=" ", flush=True)
|
||||||
|
try:
|
||||||
|
r = await client.post(
|
||||||
|
f"{OLLAMA_URL}/api/generate",
|
||||||
|
json={"model": model_name, "prompt": "hi", "stream": False},
|
||||||
|
timeout=120,
|
||||||
|
)
|
||||||
|
r.raise_for_status()
|
||||||
|
duration = r.json().get("total_duration", 0) / 1e9
|
||||||
|
print(f"OK ({duration:.1f}s)")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"WARN: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Core benchmark logic
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def run_scenario(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
scenario: Scenario,
|
||||||
|
model: str,
|
||||||
|
iteration: int,
|
||||||
|
) -> RunResult:
|
||||||
|
"""Run a single scenario through the Tatlock API."""
|
||||||
|
payload = {
|
||||||
|
"model": "Tatlock",
|
||||||
|
"messages": [{"role": "user", "content": scenario.prompt}],
|
||||||
|
}
|
||||||
|
|
||||||
|
start = time.monotonic()
|
||||||
|
try:
|
||||||
|
r = await client.post(CHAT_URL, json=payload, timeout=120)
|
||||||
|
latency = time.monotonic() - start
|
||||||
|
|
||||||
|
if r.status_code != 200:
|
||||||
|
return RunResult(
|
||||||
|
scenario=scenario.name,
|
||||||
|
model=model,
|
||||||
|
iteration=iteration,
|
||||||
|
latency=latency,
|
||||||
|
response_text="",
|
||||||
|
has_correct_answer=False,
|
||||||
|
error=f"HTTP {r.status_code}: {r.text[:100]}",
|
||||||
|
)
|
||||||
|
|
||||||
|
data = r.json()
|
||||||
|
response_text = data["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
# Check if the response contains expected patterns
|
||||||
|
has_correct = True
|
||||||
|
if scenario.success_patterns:
|
||||||
|
has_correct = any(
|
||||||
|
p.lower() in response_text.lower()
|
||||||
|
for p in scenario.success_patterns
|
||||||
|
)
|
||||||
|
|
||||||
|
return RunResult(
|
||||||
|
scenario=scenario.name,
|
||||||
|
model=model,
|
||||||
|
iteration=iteration,
|
||||||
|
latency=latency,
|
||||||
|
response_text=response_text,
|
||||||
|
has_correct_answer=has_correct,
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
latency = time.monotonic() - start
|
||||||
|
return RunResult(
|
||||||
|
scenario=scenario.name,
|
||||||
|
model=model,
|
||||||
|
iteration=iteration,
|
||||||
|
latency=latency,
|
||||||
|
response_text="",
|
||||||
|
has_correct_answer=False,
|
||||||
|
error=str(e)[:200],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def benchmark_model(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
model_name: str,
|
||||||
|
iterations: int,
|
||||||
|
) -> ModelStats:
|
||||||
|
"""Run all scenarios for a single model."""
|
||||||
|
stats = ModelStats(model=model_name)
|
||||||
|
|
||||||
|
print(f"\n{'=' * 70}")
|
||||||
|
print(f" Model: {model_name}")
|
||||||
|
print(f"{'=' * 70}")
|
||||||
|
|
||||||
|
# Swap model in .env
|
||||||
|
swap_model_in_env(model_name)
|
||||||
|
|
||||||
|
# Warm up model in Ollama BEFORE server reload picks it up
|
||||||
|
await warm_up_ollama_model(client, model_name)
|
||||||
|
|
||||||
|
# Wait for server to reload with new model
|
||||||
|
print(" Waiting for server reload...", end=" ", flush=True)
|
||||||
|
await wait_for_server_reload(client)
|
||||||
|
print("OK")
|
||||||
|
|
||||||
|
# Run a throwaway request through the full pipeline to warm up
|
||||||
|
print(" Warming up pipeline...", end=" ", flush=True)
|
||||||
|
try:
|
||||||
|
await client.post(
|
||||||
|
CHAT_URL,
|
||||||
|
json={"model": "Tatlock", "messages": [{"role": "user", "content": "hi"}]},
|
||||||
|
timeout=120,
|
||||||
|
)
|
||||||
|
print("OK")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"WARN: {e}")
|
||||||
|
|
||||||
|
for iteration in range(iterations):
|
||||||
|
if iterations > 1:
|
||||||
|
print(f"\n --- Iteration {iteration + 1}/{iterations} ---")
|
||||||
|
|
||||||
|
for scenario in SCENARIOS:
|
||||||
|
result = await run_scenario(client, scenario, model_name, iteration)
|
||||||
|
stats.results.append(result)
|
||||||
|
|
||||||
|
# Display
|
||||||
|
if result.error:
|
||||||
|
print(
|
||||||
|
f" [ERR ] {scenario.name:30s} {result.latency:5.1f}s "
|
||||||
|
f"{result.error[:60]}"
|
||||||
|
)
|
||||||
|
elif result.has_correct_answer:
|
||||||
|
preview = result.response_text[:60].replace("\n", " ")
|
||||||
|
print(f" [OK ] {scenario.name:30s} {result.latency:5.1f}s {preview}")
|
||||||
|
else:
|
||||||
|
preview = result.response_text[:60].replace("\n", " ")
|
||||||
|
print(f" [MISS] {scenario.name:30s} {result.latency:5.1f}s {preview}")
|
||||||
|
|
||||||
|
return stats
|
||||||
|
|
||||||
|
|
||||||
|
def print_comparison(all_stats: list[ModelStats]):
|
||||||
|
"""Print side-by-side comparison table."""
|
||||||
|
print("\n" + "=" * 80)
|
||||||
|
print(" COMPARISON SUMMARY")
|
||||||
|
print("=" * 80)
|
||||||
|
|
||||||
|
col_width = max(len(s.model) for s in all_stats) + 2
|
||||||
|
label_width = 32
|
||||||
|
|
||||||
|
header = f"{'Metric':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
header += f" {s.model:>{col_width}}"
|
||||||
|
print(f"\n{header}")
|
||||||
|
print("-" * (label_width + (col_width + 2) * len(all_stats)))
|
||||||
|
|
||||||
|
# Answer accuracy
|
||||||
|
row = f"{'Correct answer rate':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.accuracy:>{col_width - 1}.1f}%"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
# Latency
|
||||||
|
row = f"{'Avg latency':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.avg_latency:>{col_width - 1}.1f}s"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
row = f"{'P95 latency':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.p95_latency:>{col_width - 1}.1f}s"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
row = f"{'Max latency':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.max_latency:>{col_width - 1}.1f}s"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
# Errors
|
||||||
|
row = f"{'Errors':<{label_width}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.errors:>{col_width}}"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
# Per-category
|
||||||
|
categories = sorted(set(sc.category for sc in SCENARIOS))
|
||||||
|
print(f"\n{'Per-category accuracy':<{label_width}}")
|
||||||
|
print("-" * (label_width + (col_width + 2) * len(all_stats)))
|
||||||
|
for cat in categories:
|
||||||
|
row = f" {cat:<{label_width - 2}}"
|
||||||
|
for s in all_stats:
|
||||||
|
row += f" {s.category_accuracy(cat):>{col_width - 1}.1f}%"
|
||||||
|
print(row)
|
||||||
|
|
||||||
|
# Mismatches
|
||||||
|
print(f"\n{'Missed answers':<50}")
|
||||||
|
print("-" * 80)
|
||||||
|
any_miss = False
|
||||||
|
for scenario in SCENARIOS:
|
||||||
|
misses = []
|
||||||
|
for s in all_stats:
|
||||||
|
sc_results = [r for r in s.results if r.scenario == scenario.name]
|
||||||
|
fails = [r for r in sc_results if not r.has_correct_answer and not r.error]
|
||||||
|
if fails:
|
||||||
|
preview = fails[0].response_text[:50].replace("\n", " ")
|
||||||
|
misses.append(f"{s.model}: \"{preview}\"")
|
||||||
|
if misses:
|
||||||
|
any_miss = True
|
||||||
|
print(f" {scenario.name}")
|
||||||
|
for m in misses:
|
||||||
|
print(f" {m}")
|
||||||
|
|
||||||
|
if not any_miss:
|
||||||
|
print(" (none)")
|
||||||
|
|
||||||
|
print("\n" + "=" * 80)
|
||||||
|
|
||||||
|
|
||||||
|
def save_results(all_stats: list[ModelStats], output_path: Path):
|
||||||
|
"""Save detailed results to JSON."""
|
||||||
|
data = {}
|
||||||
|
for stats in all_stats:
|
||||||
|
data[stats.model] = {
|
||||||
|
"summary": {
|
||||||
|
"accuracy": stats.accuracy,
|
||||||
|
"avg_latency": round(stats.avg_latency, 2),
|
||||||
|
"p95_latency": round(stats.p95_latency, 2),
|
||||||
|
"max_latency": round(stats.max_latency, 2),
|
||||||
|
"errors": stats.errors,
|
||||||
|
"total_runs": stats.total,
|
||||||
|
},
|
||||||
|
"runs": [
|
||||||
|
{
|
||||||
|
"scenario": r.scenario,
|
||||||
|
"iteration": r.iteration,
|
||||||
|
"latency": round(r.latency, 3),
|
||||||
|
"has_correct_answer": r.has_correct_answer,
|
||||||
|
"response_text": r.response_text,
|
||||||
|
"error": r.error,
|
||||||
|
}
|
||||||
|
for r in stats.results
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
output_path.write_text(json.dumps(data, indent=2))
|
||||||
|
print(f"\nDetailed results saved to: {output_path}")
|
||||||
|
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Benchmark tool calling across Ollama models via Tatlock API")
|
||||||
|
parser.add_argument(
|
||||||
|
"--iterations", type=int, default=1,
|
||||||
|
help="Iterations per model (default: 1)",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--models", type=str, default=",".join(DEFAULT_MODELS),
|
||||||
|
help=f"Comma-separated models (default: {','.join(DEFAULT_MODELS)})",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--output", type=str, default="logs/benchmark_results.json",
|
||||||
|
help="JSON output path (default: logs/benchmark_results.json)",
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
models = [m.strip() for m in args.models.split(",")]
|
||||||
|
|
||||||
|
# Verify server is running
|
||||||
|
async with httpx.AsyncClient() as client:
|
||||||
|
try:
|
||||||
|
r = await client.get(HEALTH_URL, timeout=5)
|
||||||
|
r.raise_for_status()
|
||||||
|
print("Server is running.")
|
||||||
|
except Exception:
|
||||||
|
print("ERROR: Server not running. Start it with ./wakeup.sh first.")
|
||||||
|
return
|
||||||
|
|
||||||
|
print("=" * 70)
|
||||||
|
print(" Tool Calling Benchmark (via Tatlock API)")
|
||||||
|
print("=" * 70)
|
||||||
|
print(f" Models: {', '.join(models)}")
|
||||||
|
print(f" Scenarios: {len(SCENARIOS)}")
|
||||||
|
print(f" Iterations: {args.iterations}")
|
||||||
|
print(f" Total runs: {len(SCENARIOS) * args.iterations * len(models)}")
|
||||||
|
|
||||||
|
# Remember original model to restore after benchmark
|
||||||
|
original_env = ENV_PATH.read_text()
|
||||||
|
|
||||||
|
all_stats = []
|
||||||
|
# Both restores must survive a crash or an interrupt. The .env one especially:
|
||||||
|
# this script rewrites OLLAMA_DEFAULT_MODEL and lets uvicorn reload onto it,
|
||||||
|
# so bailing out mid-run used to leave the *running server* pointed at the
|
||||||
|
# benchmark model — and DEFAULT_MODELS starts at mistral-nemo-large, the 9.2G
|
||||||
|
# model implicated in the 2026-08-07 VRAM outage.
|
||||||
|
install_sigterm_handler()
|
||||||
|
try:
|
||||||
|
with residency_guard(models_used=models):
|
||||||
|
async with httpx.AsyncClient() as client:
|
||||||
|
for model in models:
|
||||||
|
stats = await benchmark_model(client, model, args.iterations)
|
||||||
|
all_stats.append(stats)
|
||||||
|
finally:
|
||||||
|
ENV_PATH.write_text(original_env)
|
||||||
|
print("\n .env restored to original")
|
||||||
|
|
||||||
|
print_comparison(all_stats)
|
||||||
|
save_results(all_stats, Path(args.output))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
asyncio.run(main())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
# .env and GPU residency are both restored by now; do not bury that
|
||||||
|
# output under a traceback.
|
||||||
|
print("\ninterrupted", file=sys.stderr)
|
||||||
|
raise SystemExit(130) from None
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
"""
|
||||||
|
Labelled queries for the Steward routing benchmark.
|
||||||
|
|
||||||
|
Each fixture carries both `expect` and `forbid`:
|
||||||
|
|
||||||
|
expect capabilities that must appear. Missing one is under-routing — the
|
||||||
|
Butler answers without a tool it needed.
|
||||||
|
forbid capabilities that must not appear. Over-routing is not cosmetic: a
|
||||||
|
spurious librarian is a real multi-second web call, and a spurious
|
||||||
|
housekeeper can actuate hardware.
|
||||||
|
|
||||||
|
`forbid` matters more than `expect` here, because over-recommendation is the
|
||||||
|
predicted failure when model thinking is disabled and the Steward has less room
|
||||||
|
to discriminate.
|
||||||
|
|
||||||
|
The `adversarial` group deserves explanation. Until 2026-08-08 the extractor
|
||||||
|
substring-matched capability *domains* across the Steward's whole response, so
|
||||||
|
ordinary English in its REASON line selected agents: "description" contains the
|
||||||
|
housekeeper domain "script", "acknowledge" contains "knowledge" and "know",
|
||||||
|
"economy" contains the biographer domain "my". Those queries invite exactly that
|
||||||
|
vocabulary. They now serve as an end-to-end regression: routing must depend on
|
||||||
|
what the Steward *decided*, not on the words it happened to use while explaining.
|
||||||
|
|
||||||
|
Expectations follow the routing rules stated in the Steward prompt itself
|
||||||
|
(src/agents/steward/agent.py), not on what a capability could plausibly cover.
|
||||||
|
"""
|
||||||
|
|
||||||
|
CORE = "tatlock_core"
|
||||||
|
LIB = "librarian"
|
||||||
|
BIO = "biographer"
|
||||||
|
HOUSE = "housekeeper"
|
||||||
|
ALL = [CORE, LIB, BIO, HOUSE]
|
||||||
|
|
||||||
|
|
||||||
|
def _others(*keep: str) -> list[str]:
|
||||||
|
return [c for c in ALL if c not in keep]
|
||||||
|
|
||||||
|
|
||||||
|
FIXTURES: list[dict] = [
|
||||||
|
# --- arithmetic and computation -> tatlock_core --------------------------
|
||||||
|
{"id": "math_add", "group": "math", "query": "What is 61 plus 12?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
{"id": "math_percent", "group": "math", "query": "What is 15% of 240?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
{"id": "math_compound", "group": "math", "query": "If I save 200 a month for 3 years, how much is that?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
{"id": "math_sqrt", "group": "math", "query": "What is the square root of 1764?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
|
||||||
|
# --- date and time -> tatlock_core ---------------------------------------
|
||||||
|
{"id": "time_now", "group": "datetime", "query": "What time is it?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
{"id": "time_date", "group": "datetime", "query": "What is today's date?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
{"id": "time_delta", "group": "datetime", "query": "How many days until Christmas?",
|
||||||
|
"expect": [CORE], "forbid": _others(CORE)},
|
||||||
|
|
||||||
|
# --- personal memory -> biographer ---------------------------------------
|
||||||
|
{"id": "bio_location", "group": "biographer", "query": "Where do I live?",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
{"id": "bio_name", "group": "biographer", "query": "What's my name?",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
{"id": "bio_car", "group": "biographer", "query": "What car do I drive?",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
{"id": "bio_store", "group": "biographer", "query": "Remember that I prefer my coffee black.",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
{"id": "bio_list", "group": "biographer", "query": "What do you know about me?",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
{"id": "bio_forget", "group": "biographer", "query": "Forget my old address.",
|
||||||
|
"expect": [BIO], "forbid": [LIB, HOUSE]},
|
||||||
|
|
||||||
|
# --- research and current information -> librarian ------------------------
|
||||||
|
{"id": "lib_weather", "group": "librarian", "query": "What's the weather in Rotterdam tomorrow?",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE]},
|
||||||
|
{"id": "lib_news", "group": "librarian", "query": "What's in the news today?",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||||
|
{"id": "lib_url", "group": "librarian", "query": "Read https://example.com/article and summarise it.",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||||
|
{"id": "lib_research", "group": "librarian", "query": "Research how tidal power stations work.",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||||
|
{"id": "lib_wiki_create", "group": "librarian", "query": "Create a wiki page about our network topology.",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE, BIO]},
|
||||||
|
|
||||||
|
# --- home automation -> housekeeper --------------------------------------
|
||||||
|
{"id": "house_lights_on", "group": "housekeeper", "query": "Turn on the kitchen lights.",
|
||||||
|
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||||
|
{"id": "house_lights_off", "group": "housekeeper", "query": "Switch off all the lights downstairs.",
|
||||||
|
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||||
|
{"id": "house_thermostat", "group": "housekeeper", "query": "Set the thermostat to 20 degrees.",
|
||||||
|
"expect": [HOUSE], "forbid": [LIB, BIO]},
|
||||||
|
{"id": "house_blinds", "group": "housekeeper", "query": "Close the blinds in the living room.",
|
||||||
|
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||||
|
|
||||||
|
# --- conversational -> nothing at all -------------------------------------
|
||||||
|
# The expensive failure mode: a greeting that triggers a web search.
|
||||||
|
{"id": "chat_greeting", "group": "conversational", "query": "Hello!",
|
||||||
|
"expect": [], "forbid": ALL},
|
||||||
|
{"id": "chat_thanks", "group": "conversational", "query": "Thanks, that's helpful.",
|
||||||
|
"expect": [], "forbid": ALL},
|
||||||
|
{"id": "chat_joke", "group": "conversational", "query": "Tell me a joke.",
|
||||||
|
"expect": [], "forbid": ALL},
|
||||||
|
{"id": "chat_howareyou", "group": "conversational", "query": "How are you doing today?",
|
||||||
|
"expect": [], "forbid": ALL},
|
||||||
|
{"id": "chat_prior_turn", "group": "conversational", "query": "What did I just say?",
|
||||||
|
"expect": [], "forbid": ALL},
|
||||||
|
|
||||||
|
# --- genuinely multi-capability -------------------------------------------
|
||||||
|
{"id": "multi_weather_home", "group": "multi",
|
||||||
|
"query": "What's the weather here, and remember that I like it warm?",
|
||||||
|
"expect": [LIB, BIO], "forbid": []},
|
||||||
|
{"id": "multi_recall_search", "group": "multi",
|
||||||
|
"query": "Look up the best route from my home address to Utrecht.",
|
||||||
|
"expect": [BIO, LIB], "forbid": []},
|
||||||
|
{"id": "multi_math_memory", "group": "multi",
|
||||||
|
"query": "Remember that my budget is 500 euro, then work out 12% of it.",
|
||||||
|
"expect": [BIO, CORE], "forbid": [LIB, HOUSE]},
|
||||||
|
|
||||||
|
# --- adversarial: vocabulary that used to select agents by substring ------
|
||||||
|
# "temperature" is a housekeeper domain, but this is a unit conversion.
|
||||||
|
{"id": "adv_temperature", "group": "adversarial", "query": "Convert 98.6 Fahrenheit to Celsius.",
|
||||||
|
"expect": [CORE], "forbid": [HOUSE, LIB, BIO]},
|
||||||
|
# "description" contains "script"; "discover" contains "cover".
|
||||||
|
{"id": "adv_description", "group": "adversarial",
|
||||||
|
"query": "Give me a short description of what 17 times 23 comes to.",
|
||||||
|
"expect": [CORE], "forbid": [HOUSE, LIB]},
|
||||||
|
# "acknowledge" contains "knowledge" and "know".
|
||||||
|
{"id": "adv_acknowledge", "group": "adversarial",
|
||||||
|
"query": "Just acknowledge this and add 5 and 6 for me.",
|
||||||
|
"expect": [CORE], "forbid": [LIB, BIO]},
|
||||||
|
# "my" appears inside "economy".
|
||||||
|
{"id": "adv_economy", "group": "adversarial",
|
||||||
|
"query": "How many zeros are in one trillion?",
|
||||||
|
"expect": [CORE], "forbid": [BIO, HOUSE]},
|
||||||
|
# "fan" inside "fantastic"; also a climate word without a home-control intent.
|
||||||
|
{"id": "adv_fantastic", "group": "adversarial",
|
||||||
|
"query": "That's fantastic. What is 8 squared?",
|
||||||
|
"expect": [CORE], "forbid": [HOUSE, LIB]},
|
||||||
|
# "home" without any actuation intent.
|
||||||
|
{"id": "adv_home_word", "group": "adversarial", "query": "What time do I usually get home?",
|
||||||
|
"expect": [BIO], "forbid": [HOUSE]},
|
||||||
|
# "search" as ordinary English, not a web-search request.
|
||||||
|
{"id": "adv_search_word", "group": "adversarial",
|
||||||
|
"query": "No need to search anything, just tell me what 9 times 9 is.",
|
||||||
|
"expect": [CORE], "forbid": [LIB]},
|
||||||
|
# "create"/"write" are librarian domains but this is conversational.
|
||||||
|
{"id": "adv_write_word", "group": "adversarial", "query": "Can you write that more simply?",
|
||||||
|
"expect": [], "forbid": [LIB, HOUSE]},
|
||||||
|
|
||||||
|
# --- mutating intents: routing only, nothing is ever executed -------------
|
||||||
|
{"id": "mutate_wiki_update", "group": "mutating", "query": "Update the dossier page with today's findings.",
|
||||||
|
"expect": [LIB], "forbid": [HOUSE, CORE]},
|
||||||
|
{"id": "mutate_scene", "group": "mutating", "query": "Run the movie night scene.",
|
||||||
|
"expect": [HOUSE], "forbid": [LIB, BIO, CORE]},
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
GROUPS = sorted({f["group"] for f in FIXTURES})
|
||||||
|
|
||||||
|
assert len({f["id"] for f in FIXTURES}) == len(FIXTURES), "duplicate fixture id"
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
"""
|
||||||
|
Guard production's GPU residency across a benchmark run.
|
||||||
|
|
||||||
|
Benchmarks swap models on the card production is serving from. Ollama evicts to
|
||||||
|
make room, so a run leaves its own models resident and the production one gone:
|
||||||
|
the next voice turn pays a ~36s cold load, and the pin that prevented it is
|
||||||
|
silently lost. That happened on 2026-08-08 — a routing benchmark evicted
|
||||||
|
gemma4:e2b and left gemma4:e4b behind, and only the monitoring noticing
|
||||||
|
`unexpected_models` caught it.
|
||||||
|
|
||||||
|
Snapshot before, restore after, and wire the restore to SIGTERM as well as the
|
||||||
|
normal path. Python runs `finally` for SIGINT, which arrives as
|
||||||
|
KeyboardInterrupt, but the default SIGTERM action terminates outright — so
|
||||||
|
`timeout`, a systemd stop or a plain `kill` would skip the guard entirely.
|
||||||
|
|
||||||
|
from scripts.ollama_residency import residency_guard, install_sigterm_handler
|
||||||
|
|
||||||
|
install_sigterm_handler()
|
||||||
|
with residency_guard(models_used=["gemma4:e4b"]):
|
||||||
|
...
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import signal
|
||||||
|
from collections.abc import Iterator
|
||||||
|
from contextlib import contextmanager
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
OLLAMA_URL = "http://localhost:11434"
|
||||||
|
|
||||||
|
# keep_alive:-1 yields a year-2318 expiry, so "pinned" is simply "expires more
|
||||||
|
# than a day out". Matches check-ai-pipeline.sh in system-admin-toj.
|
||||||
|
PINNED_THRESHOLD_SECONDS = 86400
|
||||||
|
|
||||||
|
|
||||||
|
def install_sigterm_handler() -> None:
|
||||||
|
"""Make SIGTERM raise, so `finally` blocks and context managers still run."""
|
||||||
|
def _raise(signum, _frame):
|
||||||
|
raise KeyboardInterrupt(f"signal {signum}")
|
||||||
|
|
||||||
|
signal.signal(signal.SIGTERM, _raise)
|
||||||
|
|
||||||
|
|
||||||
|
def snapshot_residency(client: httpx.Client | None = None) -> dict[str, bool]:
|
||||||
|
"""Resident models mapped to whether each is pinned."""
|
||||||
|
owns = client is None
|
||||||
|
client = client or httpx.Client(timeout=30)
|
||||||
|
try:
|
||||||
|
data = client.get(f"{OLLAMA_URL}/api/ps", timeout=10).json()
|
||||||
|
except Exception: # noqa: BLE001 - a missing snapshot must not abort the run
|
||||||
|
return {}
|
||||||
|
finally:
|
||||||
|
if owns:
|
||||||
|
client.close()
|
||||||
|
|
||||||
|
resident: dict[str, bool] = {}
|
||||||
|
now = datetime.now(UTC)
|
||||||
|
for model in data.get("models", []):
|
||||||
|
pinned = False
|
||||||
|
try:
|
||||||
|
expires = datetime.fromisoformat(model.get("expires_at", "").replace("Z", "+00:00"))
|
||||||
|
pinned = (expires - now).total_seconds() > PINNED_THRESHOLD_SECONDS
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
resident[model["name"]] = pinned
|
||||||
|
return resident
|
||||||
|
|
||||||
|
|
||||||
|
def set_keep_alive(model: str, keep_alive: Any, client: httpx.Client | None = None) -> bool:
|
||||||
|
"""Load, unload or pin a model. Embedding models reject /api/generate."""
|
||||||
|
owns = client is None
|
||||||
|
client = client or httpx.Client(timeout=180)
|
||||||
|
payload = {"model": model, "keep_alive": keep_alive}
|
||||||
|
try:
|
||||||
|
for endpoint in ("generate", "embed"):
|
||||||
|
try:
|
||||||
|
response = client.post(f"{OLLAMA_URL}/api/{endpoint}", json=payload, timeout=180)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return False
|
||||||
|
if response.status_code == 200:
|
||||||
|
return True
|
||||||
|
if response.status_code == 400 and "does not support generate" in response.text:
|
||||||
|
continue # embedding-only model; try /api/embed
|
||||||
|
return False
|
||||||
|
return False
|
||||||
|
finally:
|
||||||
|
if owns:
|
||||||
|
client.close()
|
||||||
|
|
||||||
|
|
||||||
|
def restore_residency(before: dict[str, bool], used: list[str]) -> None:
|
||||||
|
"""Evict what the benchmark loaded, then re-pin what was pinned before."""
|
||||||
|
base = {name.split(":")[0] for name in before}
|
||||||
|
with httpx.Client(timeout=180) as client:
|
||||||
|
for model in used:
|
||||||
|
if model not in before and model.split(":")[0] not in base:
|
||||||
|
print(f" residency: unloading benchmark model {model}")
|
||||||
|
set_keep_alive(model, 0, client)
|
||||||
|
for name, pinned in before.items():
|
||||||
|
if not pinned:
|
||||||
|
continue
|
||||||
|
ok = set_keep_alive(name, -1, client)
|
||||||
|
print(f" residency: re-pinned {name}" if ok
|
||||||
|
else f" residency: FAILED to re-pin {name} -- run warmup-ollama.sh")
|
||||||
|
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def residency_guard(models_used: list[str]) -> Iterator[dict[str, bool]]:
|
||||||
|
"""Snapshot residency on entry, restore it on exit however that happens."""
|
||||||
|
before = snapshot_residency()
|
||||||
|
pinned = [n for n, p in before.items() if p]
|
||||||
|
print(f" residency: resident before {sorted(before)}"
|
||||||
|
f"{f' (pinned: {pinned})' if pinned else ''}")
|
||||||
|
try:
|
||||||
|
yield before
|
||||||
|
finally:
|
||||||
|
print(" residency: restoring ...")
|
||||||
|
restore_residency(before, models_used)
|
||||||
Executable
+141
@@ -0,0 +1,141 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Housekeeper Room Group Detection Test Suite
|
||||||
|
# Verifies room groups are controlled by checking actual state changes
|
||||||
|
|
||||||
|
API_URL="http://localhost:8777/v1/chat/completions"
|
||||||
|
CORE_API="http://localhost:8083"
|
||||||
|
RESULTS_FILE="/tmp/housekeeper_test_results.txt"
|
||||||
|
|
||||||
|
GREEN='\033[0;32m'
|
||||||
|
RED='\033[0;31m'
|
||||||
|
YELLOW='\033[1;33m'
|
||||||
|
NC='\033[0m'
|
||||||
|
|
||||||
|
get_state() {
|
||||||
|
curl -s "$CORE_API/housekeeping/devices/$1" 2>/dev/null | jq -r '.state' 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
echo "=========================================="
|
||||||
|
echo "Housekeeper Room Group Test Suite"
|
||||||
|
echo "=========================================="
|
||||||
|
echo ""
|
||||||
|
|
||||||
|
> "$RESULTS_FILE"
|
||||||
|
|
||||||
|
run_toggle_test() {
|
||||||
|
local test_num=$1
|
||||||
|
local room=$2
|
||||||
|
local entity="light.$room"
|
||||||
|
local prompt_room="${room//_/ }"
|
||||||
|
|
||||||
|
printf "Test %2d: Toggle %-12s lights ... " "$test_num" "$prompt_room"
|
||||||
|
|
||||||
|
local before=$(get_state "$entity")
|
||||||
|
if [ -z "$before" ] || [ "$before" = "null" ]; then
|
||||||
|
echo -e "${YELLOW}SKIP${NC} (cannot get state)"
|
||||||
|
echo "SKIP|$test_num|Toggle $room|error" >> "$RESULTS_FILE"
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
|
curl -s -X POST "$API_URL" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d "{\"model\": \"tatlock\", \"messages\": [{\"role\": \"user\", \"content\": \"Toggle the $prompt_room lights\"}]}" > /dev/null
|
||||||
|
|
||||||
|
sleep 4
|
||||||
|
|
||||||
|
local after=$(get_state "$entity")
|
||||||
|
|
||||||
|
if [ "$before" != "$after" ]; then
|
||||||
|
echo -e "${GREEN}PASS${NC} ($before -> $after)"
|
||||||
|
echo "PASS|$test_num|Toggle $room|$before->$after" >> "$RESULTS_FILE"
|
||||||
|
else
|
||||||
|
echo -e "${RED}FAIL${NC} (state unchanged: $before)"
|
||||||
|
echo "FAIL|$test_num|Toggle $room|unchanged:$before" >> "$RESULTS_FILE"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
run_onoff_test() {
|
||||||
|
local test_num=$1
|
||||||
|
local room=$2
|
||||||
|
local action=$3
|
||||||
|
local expected_state=$4
|
||||||
|
# Entity uses underscore, prompt uses space
|
||||||
|
local entity="light.${room//_/ }"
|
||||||
|
entity="light.$room"
|
||||||
|
local prompt_room="${room//_/ }"
|
||||||
|
|
||||||
|
printf "Test %2d: %-8s %-12s lights ... " "$test_num" "$action" "$prompt_room"
|
||||||
|
|
||||||
|
curl -s -X POST "$API_URL" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d "{\"model\": \"tatlock\", \"messages\": [{\"role\": \"user\", \"content\": \"$action the $prompt_room lights\"}]}" > /dev/null
|
||||||
|
|
||||||
|
sleep 4
|
||||||
|
|
||||||
|
local after=$(get_state "$entity")
|
||||||
|
|
||||||
|
if [ "$after" = "$expected_state" ]; then
|
||||||
|
echo -e "${GREEN}PASS${NC} ($after)"
|
||||||
|
echo "PASS|$test_num|$action $room|$after" >> "$RESULTS_FILE"
|
||||||
|
else
|
||||||
|
echo -e "${RED}FAIL${NC} (got $after, expected $expected_state)"
|
||||||
|
echo "FAIL|$test_num|$action $room|got:$after,expected:$expected_state" >> "$RESULTS_FILE"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
echo "Running tests (~4s each)..."
|
||||||
|
echo ""
|
||||||
|
|
||||||
|
# Study tests
|
||||||
|
run_onoff_test 1 "study" "Turn off" "off"
|
||||||
|
run_onoff_test 2 "study" "Turn on" "on"
|
||||||
|
run_toggle_test 3 "study"
|
||||||
|
|
||||||
|
# Kitchen tests
|
||||||
|
run_onoff_test 4 "kitchen" "Turn off" "off"
|
||||||
|
run_onoff_test 5 "kitchen" "Turn on" "on"
|
||||||
|
run_toggle_test 6 "kitchen"
|
||||||
|
|
||||||
|
# Bedroom tests
|
||||||
|
run_onoff_test 7 "bedroom" "Turn off" "off"
|
||||||
|
run_onoff_test 8 "bedroom" "Turn on" "on"
|
||||||
|
|
||||||
|
# Living room tests (entity is light.living_room)
|
||||||
|
run_onoff_test 9 "living_room" "Turn off" "off"
|
||||||
|
run_onoff_test 10 "living_room" "Turn on" "on"
|
||||||
|
|
||||||
|
# Ensure all lights end up ON
|
||||||
|
echo ""
|
||||||
|
echo "Restoring all lights to ON..."
|
||||||
|
for room in "study" "kitchen" "bedroom" "living room"; do
|
||||||
|
curl -s -X POST "$API_URL" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d "{\"model\": \"tatlock\", \"messages\": [{\"role\": \"user\", \"content\": \"Turn on the $room lights\"}]}" > /dev/null
|
||||||
|
sleep 3
|
||||||
|
done
|
||||||
|
echo "Done."
|
||||||
|
|
||||||
|
echo ""
|
||||||
|
echo "=========================================="
|
||||||
|
echo "Results"
|
||||||
|
echo "=========================================="
|
||||||
|
|
||||||
|
PASS=$(grep -c "^PASS" "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||||
|
FAIL=$(grep -c "^FAIL" "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||||
|
SKIP=$(grep -c "^SKIP" "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||||
|
TOTAL=$((PASS + FAIL))
|
||||||
|
|
||||||
|
echo "Passed: $PASS"
|
||||||
|
echo "Failed: $FAIL"
|
||||||
|
echo "Skipped: $SKIP"
|
||||||
|
|
||||||
|
if [ "$TOTAL" -gt 0 ]; then
|
||||||
|
echo ""
|
||||||
|
echo "Success Rate: $((PASS * 100 / TOTAL))% ($PASS/$TOTAL)"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$FAIL" -gt 0 ]; then
|
||||||
|
echo ""
|
||||||
|
echo "Failures:"
|
||||||
|
grep "^FAIL" "$RESULTS_FILE"
|
||||||
|
fi
|
||||||
+5
-9
@@ -6,7 +6,8 @@ must implement. The interface is designed around the Responses API format.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from typing import AsyncGenerator, Any
|
from collections.abc import AsyncGenerator
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
class OutputItem:
|
class OutputItem:
|
||||||
@@ -19,12 +20,7 @@ class OutputItem:
|
|||||||
- message: Assistant response message
|
- message: Assistant response message
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(
|
def __init__(self, type: str, id: str, **kwargs: Any):
|
||||||
self,
|
|
||||||
type: str,
|
|
||||||
id: str,
|
|
||||||
**kwargs: Any
|
|
||||||
):
|
|
||||||
self.type = type
|
self.type = type
|
||||||
self.id = id
|
self.id = id
|
||||||
self.data = kwargs
|
self.data = kwargs
|
||||||
@@ -40,7 +36,7 @@ class AgentInterface(ABC):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
async def generate_response(
|
def generate_response(
|
||||||
self,
|
self,
|
||||||
messages: list[dict],
|
messages: list[dict],
|
||||||
reasoning: dict | None = None,
|
reasoning: dict | None = None,
|
||||||
@@ -48,7 +44,7 @@ class AgentInterface(ABC):
|
|||||||
temperature: float = 1.0,
|
temperature: float = 1.0,
|
||||||
max_tokens: int | None = None,
|
max_tokens: int | None = None,
|
||||||
stop: list[str] | None = None,
|
stop: list[str] | None = None,
|
||||||
**kwargs: Any
|
**kwargs: Any,
|
||||||
) -> AsyncGenerator[OutputItem, None]:
|
) -> AsyncGenerator[OutputItem, None]:
|
||||||
"""
|
"""
|
||||||
Generate streaming response as output items.
|
Generate streaming response as output items.
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
"""
|
||||||
|
The Biographer - Expert for recording and recalling the user's story.
|
||||||
|
|
||||||
|
The Biographer serves as the household's memory keeper, responsible for:
|
||||||
|
- Recording and recalling facts about the user's life
|
||||||
|
- Storing personal information, preferences, and insights
|
||||||
|
- Answering questions like "What car do I drive?", "Where do I work?"
|
||||||
|
- Managing what the household knows and remembers
|
||||||
|
|
||||||
|
For direct key-based lookups (location, timezone, preferences),
|
||||||
|
use the memory_service instead - it's faster and doesn't require LLM.
|
||||||
|
The Biographer handles semantic, fuzzy queries.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.biographer.agent import (
|
||||||
|
get_biographer_agent,
|
||||||
|
run_biographer,
|
||||||
|
run_biographer_stream,
|
||||||
|
)
|
||||||
|
from src.agents.biographer.capability import (
|
||||||
|
BIOGRAPHER_CAPABILITY,
|
||||||
|
get_biographer_capability,
|
||||||
|
register_biographer,
|
||||||
|
unregister_biographer,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"BIOGRAPHER_CAPABILITY",
|
||||||
|
"get_biographer_capability",
|
||||||
|
"get_biographer_agent",
|
||||||
|
"register_biographer",
|
||||||
|
"unregister_biographer",
|
||||||
|
"run_biographer",
|
||||||
|
"run_biographer_stream",
|
||||||
|
]
|
||||||
@@ -0,0 +1,268 @@
|
|||||||
|
"""
|
||||||
|
The Biographer - Expert for recording and recalling the user's story.
|
||||||
|
|
||||||
|
A PydanticAI agent that serves as the household's memory keeper:
|
||||||
|
- Records facts about the user's life, work, and preferences
|
||||||
|
- Recalls information semantically ("What car do I drive?")
|
||||||
|
- Manages user profile and preferences
|
||||||
|
- Forgets information when requested
|
||||||
|
"""
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from pydantic_ai import Agent
|
||||||
|
|
||||||
|
from src.agents.biographer.tools import (
|
||||||
|
forget_memory,
|
||||||
|
list_memories,
|
||||||
|
recall_semantic,
|
||||||
|
store_insight,
|
||||||
|
update_preference,
|
||||||
|
update_profile,
|
||||||
|
)
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
# The Biographer's system prompt
|
||||||
|
BIOGRAPHER_SYSTEM_PROMPT = """You are The Biographer, the household's memory keeper in the Tatlock estate.
|
||||||
|
|
||||||
|
Your role is to record, recall, and manage the story of the user's life:
|
||||||
|
- Personal facts (vehicle, pets, family members, hobbies, interests)
|
||||||
|
- Life details (employer, occupation, significant events)
|
||||||
|
- Profile information (name, location, timezone)
|
||||||
|
- Preferences (units, theme, communication style)
|
||||||
|
|
||||||
|
## Your Character
|
||||||
|
|
||||||
|
You are a discreet and attentive chronicler. Like a personal biographer who has been
|
||||||
|
with the household for years, you:
|
||||||
|
- Listen carefully and remember important details
|
||||||
|
- Recall information accurately when asked
|
||||||
|
- Never gossip or volunteer unnecessary information
|
||||||
|
- Respect privacy absolutely
|
||||||
|
- Acknowledge when you don't know something rather than guessing
|
||||||
|
|
||||||
|
## Your Tools
|
||||||
|
|
||||||
|
### Recalling the Story
|
||||||
|
- **recall_semantic**: Your primary tool for answering questions about the user
|
||||||
|
- "What car do I drive?" → searches for car-related memories
|
||||||
|
- "Where do I work?" → finds employment information
|
||||||
|
- Finds relevant memories even without exact keywords
|
||||||
|
- **list_memories**: Browse all recorded memories of a type
|
||||||
|
- Use when user asks "What do you know about me?"
|
||||||
|
- Shows everything you've recorded
|
||||||
|
|
||||||
|
### Recording New Details
|
||||||
|
- **store_insight**: Record new facts from conversation
|
||||||
|
- User says "My car is a Tesla" → store_insight("car", "Tesla Model 3")
|
||||||
|
- User says "I work at Acme" → store_insight("employer", "Acme Corp")
|
||||||
|
- Use for facts that don't fit standard profile fields
|
||||||
|
- **update_profile**: Update core biographical fields
|
||||||
|
- name, location, timezone only
|
||||||
|
- "I live in Amsterdam" → update_profile("location", "Amsterdam")
|
||||||
|
- **update_preference**: Record user preferences
|
||||||
|
- temperature_unit, distance_unit, theme, etc.
|
||||||
|
- "Use Celsius please" → update_preference("temperature_unit", "celsius")
|
||||||
|
|
||||||
|
### Managing Records
|
||||||
|
- **forget_memory**: Remove specific records
|
||||||
|
- User asks to forget something → honor immediately
|
||||||
|
- Information becomes outdated → remove it
|
||||||
|
|
||||||
|
## Guidelines
|
||||||
|
|
||||||
|
### What to Record
|
||||||
|
- Explicit statements: "I drive a Tesla", "My wife is Sarah"
|
||||||
|
- Corrections: "Actually, I moved to Berlin"
|
||||||
|
- Preferences: "I prefer metric units"
|
||||||
|
|
||||||
|
### What NOT to Record
|
||||||
|
- Sensitive data: passwords, financial details, health information
|
||||||
|
- Temporary information: "I'm tired today"
|
||||||
|
- Speculation or assumptions
|
||||||
|
|
||||||
|
### Responding to Tatlock
|
||||||
|
Your responses go to Tatlock (the butler) who synthesizes the final answer. Be:
|
||||||
|
- Direct and factual
|
||||||
|
- Clear about what you found or didn't find
|
||||||
|
- Structured for easy integration with other responses
|
||||||
|
|
||||||
|
When you don't have information:
|
||||||
|
"I have no record of the user's [topic]. Would you like me to record this information?"
|
||||||
|
|
||||||
|
When recalling:
|
||||||
|
"According to my records, [information]. This was recorded [source/when if available]."
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Lazy initialization to avoid connection issues during imports
|
||||||
|
_biographer_agent: Agent[None, str] | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _create_biographer_agent() -> Agent[None, str]:
|
||||||
|
"""Create The Biographer PydanticAI agent."""
|
||||||
|
from src.anthropic.model_selector import get_model
|
||||||
|
|
||||||
|
# Get best available model (Claude if available, else Ollama)
|
||||||
|
model = get_model()
|
||||||
|
|
||||||
|
agent: Agent[None, str] = Agent(
|
||||||
|
model=model,
|
||||||
|
system_prompt=BIOGRAPHER_SYSTEM_PROMPT,
|
||||||
|
retries=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Register recall tools
|
||||||
|
agent.tool_plain(recall_semantic)
|
||||||
|
agent.tool_plain(list_memories)
|
||||||
|
|
||||||
|
# Register recording tools
|
||||||
|
agent.tool_plain(store_insight)
|
||||||
|
agent.tool_plain(update_profile)
|
||||||
|
agent.tool_plain(update_preference)
|
||||||
|
|
||||||
|
# Register management tools
|
||||||
|
agent.tool_plain(forget_memory)
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model_info
|
||||||
|
|
||||||
|
model_info = get_model_info()
|
||||||
|
logger.info(
|
||||||
|
"biographer_agent_created",
|
||||||
|
backend=model_info["backend"],
|
||||||
|
model=model_info["model"],
|
||||||
|
tool_count=6,
|
||||||
|
)
|
||||||
|
|
||||||
|
return agent
|
||||||
|
|
||||||
|
|
||||||
|
def get_biographer_agent() -> Agent[None, str]:
|
||||||
|
"""
|
||||||
|
Get The Biographer agent instance (lazy initialization).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PydanticAI Agent configured for memory tasks
|
||||||
|
"""
|
||||||
|
global _biographer_agent
|
||||||
|
if _biographer_agent is None:
|
||||||
|
_biographer_agent = _create_biographer_agent()
|
||||||
|
return _biographer_agent
|
||||||
|
|
||||||
|
|
||||||
|
async def run_biographer(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
message_history: list[Any] | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Execute a memory task with The Biographer.
|
||||||
|
|
||||||
|
This is the main entry point for delegating memory tasks
|
||||||
|
from Tatlock or other agents.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: The memory task or question
|
||||||
|
context: Additional context from conversation
|
||||||
|
message_history: Optional conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Memory results or confirmation
|
||||||
|
|
||||||
|
Example:
|
||||||
|
result = await run_biographer(
|
||||||
|
task="What car do I drive?",
|
||||||
|
context="User is asking about their vehicle",
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
agent = get_biographer_agent()
|
||||||
|
|
||||||
|
# Build prompt with context if provided
|
||||||
|
prompt = task
|
||||||
|
if context:
|
||||||
|
prompt = f"Context: {context}\n\nTask: {task}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"biographer_task_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
has_history=bool(message_history),
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
result = await agent.run(
|
||||||
|
prompt,
|
||||||
|
message_history=message_history,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"biographer_task_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(result.output),
|
||||||
|
)
|
||||||
|
|
||||||
|
return result.output
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"biographer_task_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
return f"The Biographer encountered an error: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def run_biographer_stream(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
message_history: list[Any] | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Execute a memory task with streaming output.
|
||||||
|
|
||||||
|
Yields text deltas as The Biographer generates the response.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: The memory task or question
|
||||||
|
context: Additional context from conversation
|
||||||
|
message_history: Optional conversation history
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
str: Text deltas from the response
|
||||||
|
|
||||||
|
Example:
|
||||||
|
async for delta in run_biographer_stream("What do you know about me?"):
|
||||||
|
print(delta, end="", flush=True)
|
||||||
|
"""
|
||||||
|
agent = get_biographer_agent()
|
||||||
|
|
||||||
|
# Build prompt with context if provided
|
||||||
|
prompt = task
|
||||||
|
if context:
|
||||||
|
prompt = f"Context: {context}\n\nTask: {task}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"biographer_stream_started",
|
||||||
|
task=task[:100],
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with agent.run_stream(
|
||||||
|
prompt,
|
||||||
|
message_history=message_history,
|
||||||
|
) as response:
|
||||||
|
async for delta in response.stream_text(delta=True):
|
||||||
|
yield delta
|
||||||
|
|
||||||
|
logger.info("biographer_stream_completed", task=task[:50])
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"biographer_stream_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
yield f"\n\nThe Biographer encountered an error: {str(e)}"
|
||||||
@@ -0,0 +1,89 @@
|
|||||||
|
"""
|
||||||
|
Biographer capability registration for the Household Registry.
|
||||||
|
|
||||||
|
Defines The Biographer's capabilities and registers it as a
|
||||||
|
household member for coordination by the Steward and Tatlock.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.biographer.agent import get_biographer_agent
|
||||||
|
from src.agents.biographer.tools import BIOGRAPHER_TOOLS
|
||||||
|
from src.core.household_registry import (
|
||||||
|
HouseholdCapability,
|
||||||
|
get_household_registry,
|
||||||
|
)
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# The Biographer's capability summary for Steward coordination
|
||||||
|
BIOGRAPHER_CAPABILITY = HouseholdCapability(
|
||||||
|
name="biographer",
|
||||||
|
role="The Biographer",
|
||||||
|
category="context",
|
||||||
|
description=(
|
||||||
|
"Memory keeper for the user's story: can RECALL personal facts "
|
||||||
|
"(car, job, family, pets), RECORD new information learned from "
|
||||||
|
"conversation, UPDATE profile (name, location, timezone) and "
|
||||||
|
"preferences (units, theme), and FORGET information when requested. "
|
||||||
|
"Use for: 'what car do I drive?', 'remember that I...', "
|
||||||
|
"'forget my...', 'what do you know about me?'"
|
||||||
|
),
|
||||||
|
domains=[
|
||||||
|
"remember",
|
||||||
|
"recall",
|
||||||
|
"forget",
|
||||||
|
"memory",
|
||||||
|
"preferences",
|
||||||
|
"profile",
|
||||||
|
"personal",
|
||||||
|
"know",
|
||||||
|
"about me",
|
||||||
|
"my",
|
||||||
|
],
|
||||||
|
cost="low", # Mostly vector search, minimal LLM
|
||||||
|
requires_network=False, # All local (Qdrant, Redis)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_biographer_capability() -> HouseholdCapability:
|
||||||
|
"""Get The Biographer's capability definition."""
|
||||||
|
return BIOGRAPHER_CAPABILITY
|
||||||
|
|
||||||
|
|
||||||
|
def register_biographer() -> None:
|
||||||
|
"""
|
||||||
|
Register The Biographer with the Household Registry.
|
||||||
|
|
||||||
|
This makes The Biographer available for:
|
||||||
|
- Steward recommendations (via capability summary)
|
||||||
|
- Tatlock delegation (via agent reference)
|
||||||
|
- Tool scoping (via tool list)
|
||||||
|
"""
|
||||||
|
registry = get_household_registry()
|
||||||
|
|
||||||
|
# Check if already registered
|
||||||
|
if "biographer" in registry:
|
||||||
|
logger.debug("biographer_already_registered")
|
||||||
|
return
|
||||||
|
|
||||||
|
registry.register(
|
||||||
|
name="biographer",
|
||||||
|
capability=BIOGRAPHER_CAPABILITY,
|
||||||
|
tools=BIOGRAPHER_TOOLS,
|
||||||
|
agent=get_biographer_agent(),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"biographer_registered",
|
||||||
|
role=BIOGRAPHER_CAPABILITY.role,
|
||||||
|
domains=BIOGRAPHER_CAPABILITY.domains,
|
||||||
|
tool_count=len(BIOGRAPHER_TOOLS),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def unregister_biographer() -> None:
|
||||||
|
"""Unregister The Biographer from the Household Registry."""
|
||||||
|
registry = get_household_registry()
|
||||||
|
registry.unregister("biographer")
|
||||||
|
logger.info("biographer_unregistered")
|
||||||
@@ -0,0 +1,462 @@
|
|||||||
|
"""
|
||||||
|
Biographer tools for PydanticAI agent.
|
||||||
|
|
||||||
|
These tools enable The Biographer to record and recall the user's story:
|
||||||
|
- recall_semantic: Find memories by meaning/concept
|
||||||
|
- store_insight: Record new facts about the user
|
||||||
|
- list_memories: Browse recorded memories by type
|
||||||
|
- forget_memory: Remove specific memories
|
||||||
|
|
||||||
|
For direct key-based access (get/set profile, preferences),
|
||||||
|
use memory_service directly - these tools are for semantic queries.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.core.context import get_user
|
||||||
|
from src.core.embeddings import get_embedding_client
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
from src.core.memory_service import MemoryType, memory_service
|
||||||
|
from src.core.qdrant import get_qdrant_client
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Semantic Recall
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def recall_semantic(
|
||||||
|
query: str,
|
||||||
|
memory_type: str = "",
|
||||||
|
limit: int = 5,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Search memories by semantic similarity.
|
||||||
|
|
||||||
|
Use this to find memories that are conceptually related to
|
||||||
|
the query, even if exact words don't match. This is the main
|
||||||
|
tool for answering questions like "What car do I drive?" or
|
||||||
|
"What did I mention about my job?"
|
||||||
|
|
||||||
|
Args:
|
||||||
|
query: Natural language query to search for
|
||||||
|
memory_type: Optional filter: "user_profile", "preference", "learned_fact"
|
||||||
|
limit: Maximum memories to return (default: 5)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Matching memories with their content and relevance scores
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
recall_semantic("What is my car?")
|
||||||
|
recall_semantic("work preferences", memory_type="preference")
|
||||||
|
recall_semantic("family members")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
user = get_user()
|
||||||
|
embedding_client = get_embedding_client()
|
||||||
|
qdrant = get_qdrant_client()
|
||||||
|
|
||||||
|
# Generate embedding for query
|
||||||
|
query_vector = await embedding_client.embed(query)
|
||||||
|
if not query_vector:
|
||||||
|
return "Unable to process query - embedding generation failed"
|
||||||
|
|
||||||
|
# Search memories
|
||||||
|
results = await qdrant.search_memories(
|
||||||
|
user=user,
|
||||||
|
query_vector=query_vector,
|
||||||
|
limit=limit,
|
||||||
|
memory_type=memory_type if memory_type else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not results:
|
||||||
|
return f"No memories found related to '{query}'"
|
||||||
|
|
||||||
|
output_parts = [f"## Memories matching: {query}\n"]
|
||||||
|
|
||||||
|
for i, memory in enumerate(results, 1):
|
||||||
|
mem_type = memory.get("type", "unknown")
|
||||||
|
key = memory.get("key", "")
|
||||||
|
value = memory.get("value", "")
|
||||||
|
score = memory.get("score", 0.0)
|
||||||
|
source = memory.get("source", "unknown")
|
||||||
|
|
||||||
|
type_icon = {
|
||||||
|
"user_profile": "👤",
|
||||||
|
"preference": "⚙️",
|
||||||
|
"learned_fact": "💡",
|
||||||
|
}.get(mem_type, "📝")
|
||||||
|
|
||||||
|
output_parts.append(f"{i}. {type_icon} **{key}** (relevance: {score:.2f})")
|
||||||
|
output_parts.append(f" {value}")
|
||||||
|
output_parts.append(f" _Type: {mem_type}, Source: {source}_")
|
||||||
|
output_parts.append("")
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_recall_semantic",
|
||||||
|
query=query[:50],
|
||||||
|
result_count=len(results),
|
||||||
|
user=user,
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_recall_semantic_error", error=str(e), query=query[:50])
|
||||||
|
return f"Error searching memories: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Store Memory
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def store_insight(
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
importance: float = 0.5,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Store a new insight or learned fact about the user.
|
||||||
|
|
||||||
|
Use this when:
|
||||||
|
- User explicitly asks to remember something
|
||||||
|
- User shares personal information worth remembering
|
||||||
|
- You learn something from conversation that should persist
|
||||||
|
|
||||||
|
The memory will be stored with vector embedding for semantic search
|
||||||
|
and can be recalled later using recall_semantic.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Short identifier for the memory (e.g., "car", "employer", "pet")
|
||||||
|
value: The actual information to remember
|
||||||
|
importance: How important is this? 0.0 (trivial) to 1.0 (critical)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of stored memory
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
store_insight("car", "User drives a Tesla Model 3")
|
||||||
|
store_insight("employer", "Works at Acme Corp as software engineer", importance=0.8)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Auto-generate keywords from key and value
|
||||||
|
keywords = [key]
|
||||||
|
words = value.lower().split()
|
||||||
|
keywords.extend([w for w in words if len(w) > 4][:5])
|
||||||
|
|
||||||
|
success = await memory_service.store_fact(
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
keywords=keywords,
|
||||||
|
importance=importance,
|
||||||
|
source="conversation",
|
||||||
|
)
|
||||||
|
|
||||||
|
if success:
|
||||||
|
output_parts = [
|
||||||
|
"## Memory Stored",
|
||||||
|
f"**Key:** {key}",
|
||||||
|
f"**Value:** {value}",
|
||||||
|
f"**Keywords:** {', '.join(keywords)}",
|
||||||
|
f"**Importance:** {importance:.1f}",
|
||||||
|
"",
|
||||||
|
"_Memory is now searchable via semantic recall._",
|
||||||
|
]
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_store_insight",
|
||||||
|
key=key,
|
||||||
|
importance=importance,
|
||||||
|
user=get_user(),
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
else:
|
||||||
|
return f"Failed to store memory for key '{key}'"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_store_insight_error", error=str(e), key=key)
|
||||||
|
return f"Error storing memory: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def update_profile(
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Update user profile information.
|
||||||
|
|
||||||
|
Use this for core identity information:
|
||||||
|
- name, location, timezone
|
||||||
|
- language preferences
|
||||||
|
- occupation
|
||||||
|
|
||||||
|
Profile data has high importance and is used for context
|
||||||
|
by the Steward during request analysis.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Profile field (e.g., "name", "location", "timezone")
|
||||||
|
value: The value to set
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of profile update
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
update_profile("location", "Amsterdam, Netherlands")
|
||||||
|
update_profile("timezone", "Europe/Amsterdam")
|
||||||
|
update_profile("name", "John")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
success = await memory_service.set_profile(
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
keywords=[key, "profile"],
|
||||||
|
)
|
||||||
|
|
||||||
|
if success:
|
||||||
|
output_parts = [
|
||||||
|
"## Profile Updated",
|
||||||
|
f"**{key}:** {value}",
|
||||||
|
"",
|
||||||
|
"_Profile data is automatically included in context._",
|
||||||
|
]
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_update_profile",
|
||||||
|
key=key,
|
||||||
|
user=get_user(),
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
else:
|
||||||
|
return f"Failed to update profile field '{key}'"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_update_profile_error", error=str(e), key=key)
|
||||||
|
return f"Error updating profile: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def update_preference(
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Update user preferences.
|
||||||
|
|
||||||
|
Use this for settings and preferences:
|
||||||
|
- temperature_unit (celsius/fahrenheit)
|
||||||
|
- distance_unit (metric/imperial)
|
||||||
|
- theme, language, etc.
|
||||||
|
|
||||||
|
Preferences are used by agents to customize responses.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Preference name (e.g., "temperature_unit", "theme")
|
||||||
|
value: Preference value
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of preference update
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
update_preference("temperature_unit", "celsius")
|
||||||
|
update_preference("distance_unit", "metric")
|
||||||
|
update_preference("theme", "dark")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
success = await memory_service.set_preference(
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
)
|
||||||
|
|
||||||
|
if success:
|
||||||
|
output_parts = [
|
||||||
|
"## Preference Updated",
|
||||||
|
f"**{key}:** {value}",
|
||||||
|
"",
|
||||||
|
"_Preference will be applied to future responses._",
|
||||||
|
]
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_update_preference",
|
||||||
|
key=key,
|
||||||
|
user=get_user(),
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
else:
|
||||||
|
return f"Failed to update preference '{key}'"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_update_preference_error", error=str(e), key=key)
|
||||||
|
return f"Error updating preference: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# List Memories
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def list_memories(
|
||||||
|
memory_type: str = "learned_fact",
|
||||||
|
limit: int = 20,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
List stored memories of a specific type.
|
||||||
|
|
||||||
|
Use this to browse what's stored in memory without
|
||||||
|
a specific search query.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
memory_type: Type to list: "user_profile", "preference", "learned_fact"
|
||||||
|
limit: Maximum memories to return (default: 20)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of memories with their keys and values
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_memories("user_profile")
|
||||||
|
list_memories("preference")
|
||||||
|
list_memories("learned_fact", limit=10)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
user = get_user()
|
||||||
|
qdrant = get_qdrant_client()
|
||||||
|
|
||||||
|
# Convert string to MemoryType
|
||||||
|
try:
|
||||||
|
MemoryType(memory_type) # validated for its ValueError; the value is unused
|
||||||
|
except ValueError:
|
||||||
|
return f"Invalid memory type '{memory_type}'. Use: user_profile, preference, or learned_fact"
|
||||||
|
|
||||||
|
# Get all memories of type
|
||||||
|
results = qdrant._client.scroll(
|
||||||
|
collection_name=f"memories_{user}",
|
||||||
|
scroll_filter={
|
||||||
|
"must": [
|
||||||
|
{"key": "type", "match": {"value": memory_type}},
|
||||||
|
]
|
||||||
|
},
|
||||||
|
limit=limit,
|
||||||
|
with_payload=True,
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
points, _ = results
|
||||||
|
if not points:
|
||||||
|
return f"No {memory_type} memories found"
|
||||||
|
|
||||||
|
type_icon = {
|
||||||
|
"user_profile": "👤",
|
||||||
|
"preference": "⚙️",
|
||||||
|
"learned_fact": "💡",
|
||||||
|
}.get(memory_type, "📝")
|
||||||
|
|
||||||
|
output_parts = [f"## {type_icon} {memory_type.replace('_', ' ').title()} Memories\n"]
|
||||||
|
|
||||||
|
for point in points:
|
||||||
|
payload = point.payload
|
||||||
|
key = payload.get("key", "unknown")
|
||||||
|
value = payload.get("value", "")
|
||||||
|
importance = payload.get("importance", 0.5)
|
||||||
|
|
||||||
|
output_parts.append(f"- **{key}**: {value}")
|
||||||
|
if importance > 0.7:
|
||||||
|
output_parts.append(f" _(importance: {importance:.1f})_")
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_list",
|
||||||
|
memory_type=memory_type,
|
||||||
|
count=len(points),
|
||||||
|
user=user,
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_list_error", error=str(e), memory_type=memory_type)
|
||||||
|
return f"Error listing memories: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Forget Memory
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def forget_memory(
|
||||||
|
key: str,
|
||||||
|
memory_type: str = "learned_fact",
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Remove a specific memory.
|
||||||
|
|
||||||
|
Use this when:
|
||||||
|
- User asks to forget something
|
||||||
|
- Information is outdated or incorrect
|
||||||
|
- Privacy concerns
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Key of the memory to forget
|
||||||
|
memory_type: Type of memory: "user_profile", "preference", "learned_fact"
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of deletion
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
forget_memory("old_car")
|
||||||
|
forget_memory("location", memory_type="user_profile")
|
||||||
|
forget_memory("theme", memory_type="preference")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Convert string to MemoryType
|
||||||
|
try:
|
||||||
|
mem_type = MemoryType(memory_type)
|
||||||
|
except ValueError:
|
||||||
|
return f"Invalid memory type '{memory_type}'. Use: user_profile, preference, or learned_fact"
|
||||||
|
|
||||||
|
success = await memory_service.delete_memory(
|
||||||
|
key=key,
|
||||||
|
memory_type=mem_type,
|
||||||
|
)
|
||||||
|
|
||||||
|
if success:
|
||||||
|
output_parts = [
|
||||||
|
"## Memory Forgotten",
|
||||||
|
f"**Key:** {key}",
|
||||||
|
f"**Type:** {memory_type}",
|
||||||
|
"",
|
||||||
|
"_Memory has been removed._",
|
||||||
|
]
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_forget",
|
||||||
|
key=key,
|
||||||
|
memory_type=memory_type,
|
||||||
|
user=get_user(),
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
else:
|
||||||
|
return f"Memory '{key}' not found or already deleted"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_forget_error", error=str(e), key=key)
|
||||||
|
return f"Error forgetting memory: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Tool Collection for Registration
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
# All tools available to The Biographer
|
||||||
|
BIOGRAPHER_TOOLS = [
|
||||||
|
# Recall
|
||||||
|
recall_semantic,
|
||||||
|
list_memories,
|
||||||
|
# Record
|
||||||
|
store_insight,
|
||||||
|
update_profile,
|
||||||
|
update_preference,
|
||||||
|
# Manage
|
||||||
|
forget_memory,
|
||||||
|
]
|
||||||
@@ -0,0 +1,579 @@
|
|||||||
|
"""
|
||||||
|
Delegation infrastructure for expert agent calls.
|
||||||
|
|
||||||
|
Provides delegation wrappers that Tatlock uses to call expert agents.
|
||||||
|
Each wrapper encapsulates the complexity of calling an expert and
|
||||||
|
returns a structured result for synthesis.
|
||||||
|
|
||||||
|
This implements the agent-as-tool pattern recommended by PydanticAI:
|
||||||
|
agents call other agents via tool wrappers, keeping each agent focused.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from enum import Enum
|
||||||
|
|
||||||
|
from src.core.config import config
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
from src.core.tracing import SpanType, trace_span
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# Action Types for Think Slug Selection
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
class ActionType(Enum):
|
||||||
|
"""
|
||||||
|
Categories of actions for selecting appropriate think messages.
|
||||||
|
|
||||||
|
Each expert has different action types that warrant different
|
||||||
|
butler-perspective messages to the user.
|
||||||
|
"""
|
||||||
|
|
||||||
|
RETRIEVE = "retrieve" # Looking up existing information
|
||||||
|
RESEARCH = "research" # Conducting new research (web search, etc.)
|
||||||
|
CREATE = "create" # Creating new content (pages, notes)
|
||||||
|
CONTROL = "control" # Controlling devices/automations
|
||||||
|
RECORD = "record" # Recording memories/notes
|
||||||
|
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# Household Think Messages (Butler's Perspective)
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
HOUSEHOLD_THINK_MESSAGES: dict[str, dict[ActionType, dict[str, str]]] = {
|
||||||
|
# Note: No <think> wrappers needed - these go to reasoning_content field
|
||||||
|
"librarian": {
|
||||||
|
ActionType.RETRIEVE: {
|
||||||
|
"start": "Allow me to consult the archives, sir.",
|
||||||
|
"success": "The Librarian has compiled the relevant findings.",
|
||||||
|
"error": "I'm afraid the archives proved difficult to access.",
|
||||||
|
},
|
||||||
|
ActionType.RESEARCH: {
|
||||||
|
"start": "I've dispatched the Librarian to conduct some fresh research.",
|
||||||
|
"success": "The Librarian has returned with findings, sir.",
|
||||||
|
"error": "The research proved inconclusive, I'm afraid.",
|
||||||
|
},
|
||||||
|
ActionType.CREATE: {
|
||||||
|
"start": "I'm having the Librarian prepare a new entry.",
|
||||||
|
"success": "The new material has been properly catalogued, sir.",
|
||||||
|
"error": "I'm afraid there was difficulty filing the entry.",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"biographer": {
|
||||||
|
ActionType.RETRIEVE: {
|
||||||
|
"start": "Let me consult the household records.",
|
||||||
|
"success": "The Biographer has located the relevant information, sir.",
|
||||||
|
"error": "I'm unable to locate those particular records.",
|
||||||
|
},
|
||||||
|
ActionType.RECORD: {
|
||||||
|
"start": "I've asked the Biographer to take note of this, sir.",
|
||||||
|
"success": "The household records have been updated accordingly.",
|
||||||
|
"error": "I'm afraid there was difficulty recording the entry.",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"housekeeper": {
|
||||||
|
ActionType.RETRIEVE: {
|
||||||
|
"start": "Allow me to inquire with the household staff.",
|
||||||
|
"success": "The staff reports the current status, sir.",
|
||||||
|
"error": "The household staff is momentarily unavailable, I'm afraid.",
|
||||||
|
},
|
||||||
|
ActionType.CONTROL: {
|
||||||
|
"start": "I'm instructing the household staff now, sir.",
|
||||||
|
"success": "The household has been configured as requested.",
|
||||||
|
"error": "I'm afraid the staff reports an issue with that request.",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _detect_action_type(expert: str, task: str) -> ActionType:
|
||||||
|
"""
|
||||||
|
Detect action type from expert name and task description.
|
||||||
|
|
||||||
|
Used to select appropriate butler-perspective think messages.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
expert: Name of the expert (librarian, biographer, housekeeper)
|
||||||
|
task: Task description
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
ActionType: Detected action type for message selection
|
||||||
|
"""
|
||||||
|
task_lower = task.lower()
|
||||||
|
|
||||||
|
if expert == "librarian":
|
||||||
|
# Web search, URL reading = RESEARCH (fresh external data)
|
||||||
|
if any(w in task_lower for w in ["search", "find", "look up", "research"]):
|
||||||
|
if any(w in task_lower for w in ["web", "online", "internet"]):
|
||||||
|
return ActionType.RESEARCH
|
||||||
|
return ActionType.RETRIEVE
|
||||||
|
if any(w in task_lower for w in ["read", "fetch", "url", "http"]):
|
||||||
|
return ActionType.RESEARCH # Reading URLs is research
|
||||||
|
if any(w in task_lower for w in ["create", "write", "add", "make", "new"]):
|
||||||
|
return ActionType.CREATE
|
||||||
|
return ActionType.RETRIEVE
|
||||||
|
|
||||||
|
elif expert == "biographer":
|
||||||
|
if any(w in task_lower for w in ["remember", "note", "record", "save", "store"]):
|
||||||
|
return ActionType.RECORD
|
||||||
|
return ActionType.RETRIEVE
|
||||||
|
|
||||||
|
elif expert == "housekeeper":
|
||||||
|
if any(w in task_lower for w in ["turn", "set", "activate", "enable", "disable", "toggle"]):
|
||||||
|
return ActionType.CONTROL
|
||||||
|
return ActionType.RETRIEVE
|
||||||
|
|
||||||
|
return ActionType.RETRIEVE
|
||||||
|
|
||||||
|
|
||||||
|
def build_delegation_context(
|
||||||
|
conversation_history: list[dict] | None,
|
||||||
|
max_turns: int = 6,
|
||||||
|
max_chars_per_turn: int = 500,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Format the most recent conversation turns as delegation context.
|
||||||
|
|
||||||
|
Experts accept a context string but the live paths never passed the
|
||||||
|
in-scope conversation history; this trims it to the last few turns
|
||||||
|
so follow-up questions ("and what about X?") keep their referent.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
conversation_history: Prior messages as {"role", "content"} dicts
|
||||||
|
max_turns: How many trailing turns to include
|
||||||
|
max_chars_per_turn: Truncation limit per turn
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: Newline-joined "role: content" lines ("" when no history)
|
||||||
|
"""
|
||||||
|
if not conversation_history:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
lines = []
|
||||||
|
for msg in conversation_history[-max_turns:]:
|
||||||
|
if not isinstance(msg, dict):
|
||||||
|
continue
|
||||||
|
role = msg.get("role", "user")
|
||||||
|
content = msg.get("content", "")
|
||||||
|
if isinstance(content, list):
|
||||||
|
# Tolerate structured content parts
|
||||||
|
content = " ".join(
|
||||||
|
part.get("text", "") if isinstance(part, dict) else str(part) for part in content
|
||||||
|
)
|
||||||
|
content = str(content).strip()
|
||||||
|
if content:
|
||||||
|
lines.append(f"{role}: {content[:max_chars_per_turn]}")
|
||||||
|
|
||||||
|
if not lines:
|
||||||
|
return ""
|
||||||
|
return "Recent conversation:\n" + "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def get_think_message(expert: str, task: str, phase: str) -> str:
|
||||||
|
"""
|
||||||
|
Get the appropriate think message for an expert delegation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
expert: Name of the expert
|
||||||
|
task: Task description (used to detect action type)
|
||||||
|
phase: One of "start", "success", "error"
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: Butler-perspective think message
|
||||||
|
"""
|
||||||
|
action_type = _detect_action_type(expert, task)
|
||||||
|
expert_messages = HOUSEHOLD_THINK_MESSAGES.get(expert, {})
|
||||||
|
action_messages = expert_messages.get(action_type, expert_messages.get(ActionType.RETRIEVE, {}))
|
||||||
|
return action_messages.get(phase, f"Consulting {expert}...")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class DelegationTask:
|
||||||
|
"""
|
||||||
|
A task to be delegated to an expert agent.
|
||||||
|
|
||||||
|
Represents a unit of work that Tatlock delegates to a specialist.
|
||||||
|
Used for tracking and orchestration of multi-expert workflows.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
expert_name: Name of the expert agent (e.g., "librarian", "memory")
|
||||||
|
task: Clear description of what needs to be done
|
||||||
|
context: Additional context from the conversation
|
||||||
|
action: Specific action verb (create, search, update, etc.)
|
||||||
|
priority: Execution priority (lower = higher priority)
|
||||||
|
depends_on: List of task IDs this task depends on
|
||||||
|
result: Result from expert after execution
|
||||||
|
"""
|
||||||
|
|
||||||
|
expert_name: str
|
||||||
|
task: str
|
||||||
|
context: str = ""
|
||||||
|
action: str = ""
|
||||||
|
priority: int = 0
|
||||||
|
depends_on: list[str] = field(default_factory=list)
|
||||||
|
result: str | None = None
|
||||||
|
task_id: str = ""
|
||||||
|
|
||||||
|
def __post_init__(self) -> None:
|
||||||
|
"""Generate task ID if not provided."""
|
||||||
|
if not self.task_id:
|
||||||
|
import uuid
|
||||||
|
|
||||||
|
self.task_id = f"{self.expert_name}_{uuid.uuid4().hex[:8]}"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class DelegationResult:
|
||||||
|
"""
|
||||||
|
Result from an expert agent delegation.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
expert_name: Which expert handled the task
|
||||||
|
task: Original task description
|
||||||
|
success: Whether the delegation succeeded
|
||||||
|
output: Expert's response/findings. On failure this holds a
|
||||||
|
curated, user-safe butler sentence (never exception detail)
|
||||||
|
error: Short user-safe error label if failed. Exception detail
|
||||||
|
stays in the logs only
|
||||||
|
"""
|
||||||
|
|
||||||
|
expert_name: str
|
||||||
|
task: str
|
||||||
|
success: bool
|
||||||
|
output: str
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
async def delegate_to_librarian(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
) -> DelegationResult:
|
||||||
|
"""
|
||||||
|
Delegate a research or wiki task to The Librarian.
|
||||||
|
|
||||||
|
The Librarian handles:
|
||||||
|
- Wiki creation (smart_create_wiki_page for topic-based)
|
||||||
|
- Wiki updates (update_wiki_page for modifications)
|
||||||
|
- Research queries (hybrid_search for comprehensive search)
|
||||||
|
- Knowledge graph exploration
|
||||||
|
- Document lookups and semantic search
|
||||||
|
|
||||||
|
This wrapper uses run() not run_stream() to avoid Ollama's
|
||||||
|
streaming + tool call bug (PydanticAI issues #1292, #2256).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: Clear description of what needs to be done.
|
||||||
|
Include the action verb (create, search, update, etc.)
|
||||||
|
Example: "Create a wiki page about CI/CD pipelines"
|
||||||
|
Example: "Search for information about Docker networking"
|
||||||
|
context: Additional context from the user's request or
|
||||||
|
conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DelegationResult with the Librarian's findings
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> result = await delegate_to_librarian(
|
||||||
|
... task="Create a wiki page about Kubernetes deployments",
|
||||||
|
... context="User is setting up a homelab cluster",
|
||||||
|
... )
|
||||||
|
>>> if result.success:
|
||||||
|
... print(result.output)
|
||||||
|
"""
|
||||||
|
from src.agents.librarian.agent import run_librarian
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_librarian_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
)
|
||||||
|
|
||||||
|
async with trace_span(
|
||||||
|
"delegate_to_librarian",
|
||||||
|
SpanType.EXPERT,
|
||||||
|
metadata={
|
||||||
|
"expert": "librarian",
|
||||||
|
"task_preview": task[:100],
|
||||||
|
"has_context": bool(context),
|
||||||
|
},
|
||||||
|
) as span:
|
||||||
|
try:
|
||||||
|
# Use run() not run_stream() - avoids Ollama bug.
|
||||||
|
# One timeout budget for the whole delegation - covers both
|
||||||
|
# live paths (steward direct delegation and streaming), which
|
||||||
|
# previously had no cap at all (SDK default ~600s per LLM call).
|
||||||
|
output = await asyncio.wait_for(
|
||||||
|
run_librarian(task=task, context=context),
|
||||||
|
timeout=config.LIBRARIAN_TIMEOUT,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_librarian_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(output),
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = True
|
||||||
|
span.metadata["output_length"] = len(output)
|
||||||
|
span.details["task"] = task
|
||||||
|
span.details["context"] = context[:500] if context else None
|
||||||
|
span.details["result_preview"] = output[:1000]
|
||||||
|
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="librarian",
|
||||||
|
task=task,
|
||||||
|
success=True,
|
||||||
|
output=output,
|
||||||
|
)
|
||||||
|
|
||||||
|
except TimeoutError:
|
||||||
|
logger.error(
|
||||||
|
"delegation_to_librarian_timeout",
|
||||||
|
task=task[:50],
|
||||||
|
timeout_seconds=config.LIBRARIAN_TIMEOUT,
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = False
|
||||||
|
span.details["error"] = f"timed out after {config.LIBRARIAN_TIMEOUT}s"
|
||||||
|
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="librarian",
|
||||||
|
task=task,
|
||||||
|
success=False,
|
||||||
|
output=(
|
||||||
|
"I'm afraid the research took longer than expected "
|
||||||
|
"and had to be abandoned, sir."
|
||||||
|
),
|
||||||
|
error="The Librarian did not respond within the time budget.",
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"delegation_to_librarian_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = False
|
||||||
|
span.details["error"] = str(e)
|
||||||
|
|
||||||
|
# Exception detail stays in the logs; the user-facing output
|
||||||
|
# is a curated butler sentence so internals never leak into
|
||||||
|
# synthesis.
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="librarian",
|
||||||
|
task=task,
|
||||||
|
success=False,
|
||||||
|
output=get_think_message("librarian", task, "error"),
|
||||||
|
error="The Librarian was unable to complete the task.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def delegate_to_biographer(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
) -> DelegationResult:
|
||||||
|
"""
|
||||||
|
Delegate a memory task to The Biographer.
|
||||||
|
|
||||||
|
The Biographer handles:
|
||||||
|
- Semantic recall ("What car do I drive?", "What's my job?")
|
||||||
|
- Recording new facts from conversation
|
||||||
|
- Profile updates (name, location, timezone)
|
||||||
|
- Preference updates (units, theme)
|
||||||
|
- Memory management (forget, list)
|
||||||
|
|
||||||
|
For direct key-based lookups (get location, get timezone), use
|
||||||
|
memory_service directly - it's faster and doesn't require LLM.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: Clear description of what needs to be done.
|
||||||
|
Include the action verb (recall, remember, forget, etc.)
|
||||||
|
Example: "What car do I drive?"
|
||||||
|
Example: "Remember that I work at Acme Corp"
|
||||||
|
context: Additional context from the user's request or
|
||||||
|
conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DelegationResult with The Biographer's response
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> result = await delegate_to_biographer(
|
||||||
|
... task="What do you know about my preferences?",
|
||||||
|
... context="User is asking about stored information",
|
||||||
|
... )
|
||||||
|
>>> if result.success:
|
||||||
|
... print(result.output)
|
||||||
|
"""
|
||||||
|
from src.agents.biographer.agent import run_biographer
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_biographer_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
)
|
||||||
|
|
||||||
|
async with trace_span(
|
||||||
|
"delegate_to_biographer",
|
||||||
|
SpanType.EXPERT,
|
||||||
|
metadata={
|
||||||
|
"expert": "biographer",
|
||||||
|
"task_preview": task[:100],
|
||||||
|
"has_context": bool(context),
|
||||||
|
},
|
||||||
|
) as span:
|
||||||
|
try:
|
||||||
|
# Use run() not run_stream() - avoids Ollama bug
|
||||||
|
output = await run_biographer(task=task, context=context)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_biographer_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(output),
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = True
|
||||||
|
span.metadata["output_length"] = len(output)
|
||||||
|
span.details["task"] = task
|
||||||
|
span.details["context"] = context[:500] if context else None
|
||||||
|
span.details["result_preview"] = output[:1000]
|
||||||
|
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="biographer",
|
||||||
|
task=task,
|
||||||
|
success=True,
|
||||||
|
output=output,
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"delegation_to_biographer_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = False
|
||||||
|
span.details["error"] = str(e)
|
||||||
|
|
||||||
|
# Exception detail stays in the logs only.
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="biographer",
|
||||||
|
task=task,
|
||||||
|
success=False,
|
||||||
|
output=get_think_message("biographer", task, "error"),
|
||||||
|
error="The Biographer was unable to complete the task.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def delegate_to_housekeeper(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
) -> DelegationResult:
|
||||||
|
"""
|
||||||
|
Delegate a home automation task to The Housekeeper.
|
||||||
|
|
||||||
|
The Housekeeper handles:
|
||||||
|
- Device control (turn on/off, toggle, brightness, color)
|
||||||
|
- Scene activation (movie night, good morning, etc.)
|
||||||
|
- Script execution (automation sequences)
|
||||||
|
- Automation management (enable/disable rules)
|
||||||
|
- Device discovery (list devices by area/type)
|
||||||
|
- State queries (get current state, history)
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: Clear description of what needs to be done.
|
||||||
|
Include the action verb (turn on, activate, list, etc.)
|
||||||
|
Example: "Turn on the living room lights"
|
||||||
|
Example: "Activate the movie night scene"
|
||||||
|
Example: "What devices are in the bedroom?"
|
||||||
|
context: Additional context from the user's request or
|
||||||
|
conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DelegationResult with The Housekeeper's response
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> result = await delegate_to_housekeeper(
|
||||||
|
... task="Turn on the bedroom lights at 50% brightness",
|
||||||
|
... context="User is getting ready for bed",
|
||||||
|
... )
|
||||||
|
>>> if result.success:
|
||||||
|
... print(result.output)
|
||||||
|
"""
|
||||||
|
from src.agents.housekeeper.agent import run_housekeeper
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_housekeeper_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
)
|
||||||
|
|
||||||
|
async with trace_span(
|
||||||
|
"delegate_to_housekeeper",
|
||||||
|
SpanType.EXPERT,
|
||||||
|
metadata={
|
||||||
|
"expert": "housekeeper",
|
||||||
|
"task_preview": task[:100],
|
||||||
|
"has_context": bool(context),
|
||||||
|
},
|
||||||
|
) as span:
|
||||||
|
try:
|
||||||
|
# Use run() not run_stream() - avoids Ollama bug
|
||||||
|
output = await run_housekeeper(task=task, context=context)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_to_housekeeper_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(output),
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = True
|
||||||
|
span.metadata["output_length"] = len(output)
|
||||||
|
span.details["task"] = task
|
||||||
|
span.details["context"] = context[:500] if context else None
|
||||||
|
span.details["result_preview"] = output[:1000]
|
||||||
|
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="housekeeper",
|
||||||
|
task=task,
|
||||||
|
success=True,
|
||||||
|
output=output,
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"delegation_to_housekeeper_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if span:
|
||||||
|
span.metadata["success"] = False
|
||||||
|
span.details["error"] = str(e)
|
||||||
|
|
||||||
|
# Exception detail stays in the logs only.
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name="housekeeper",
|
||||||
|
task=task,
|
||||||
|
success=False,
|
||||||
|
output=get_think_message("housekeeper", task, "error"),
|
||||||
|
error="The Housekeeper was unable to complete the task.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# Future expert delegation wrappers will be added here:
|
||||||
|
# - delegate_to_developer(task, context) -> DelegationResult
|
||||||
|
# - delegate_to_secretary(task, context) -> DelegationResult
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
"""
|
||||||
|
The Housekeeper - Home Automation Agent.
|
||||||
|
|
||||||
|
Provides home automation capabilities through the core-api service,
|
||||||
|
which wraps the Home Assistant REST API into LLM-friendly endpoints.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.housekeeper.agent import run_housekeeper, run_housekeeper_stream
|
||||||
|
from src.agents.housekeeper.capability import (
|
||||||
|
HOUSEKEEPER_CAPABILITY,
|
||||||
|
register_housekeeper,
|
||||||
|
)
|
||||||
|
from src.agents.housekeeper.client import CoreAPIClient, get_core_api_client
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
# Agent entry points
|
||||||
|
"run_housekeeper",
|
||||||
|
"run_housekeeper_stream",
|
||||||
|
# Capability
|
||||||
|
"HOUSEKEEPER_CAPABILITY",
|
||||||
|
"register_housekeeper",
|
||||||
|
# Client
|
||||||
|
"CoreAPIClient",
|
||||||
|
"get_core_api_client",
|
||||||
|
]
|
||||||
@@ -0,0 +1,290 @@
|
|||||||
|
"""
|
||||||
|
The Housekeeper - Expert agent for home automation.
|
||||||
|
|
||||||
|
A PydanticAI agent that provides home automation capabilities through
|
||||||
|
the core-api service, which wraps Home Assistant REST API, offering:
|
||||||
|
- Device discovery and control
|
||||||
|
- Scene activation
|
||||||
|
- Script execution
|
||||||
|
- Automation management
|
||||||
|
"""
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from pydantic_ai import Agent
|
||||||
|
|
||||||
|
from src.agents.housekeeper.tools import (
|
||||||
|
activate_scene,
|
||||||
|
get_device_state,
|
||||||
|
get_history,
|
||||||
|
list_areas,
|
||||||
|
list_automations,
|
||||||
|
list_devices,
|
||||||
|
list_scenes,
|
||||||
|
list_scripts,
|
||||||
|
run_script,
|
||||||
|
toggle,
|
||||||
|
toggle_automation,
|
||||||
|
turn_off,
|
||||||
|
turn_on,
|
||||||
|
)
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
# Housekeeper system prompt - Optimized for Mistral-Nemo function calling
|
||||||
|
HOUSEKEEPER_SYSTEM_PROMPT = """You are a strictly tool-based home automation assistant.
|
||||||
|
|
||||||
|
## CRITICAL: You Have NO Internal Knowledge
|
||||||
|
|
||||||
|
You do NOT know what devices exist. You do NOT know any entity IDs.
|
||||||
|
Entity IDs are different in every installation. You MUST discover them using tools.
|
||||||
|
|
||||||
|
## Entity ID Format
|
||||||
|
|
||||||
|
Entity IDs follow the format: `domain.name`
|
||||||
|
Examples: `light.kitchen`, `light.study_main`, `switch.coffee_maker`
|
||||||
|
|
||||||
|
The `entity_id` parameter MUST be the COMPLETE value including the domain prefix.
|
||||||
|
WRONG: `entity_id="kitchen"`
|
||||||
|
RIGHT: `entity_id="light.kitchen"`
|
||||||
|
|
||||||
|
## Step-by-Step Process (ALWAYS FOLLOW)
|
||||||
|
|
||||||
|
When asked to control devices in a room:
|
||||||
|
|
||||||
|
1. THINK: What domain? (light, switch, climate, etc.)
|
||||||
|
2. CALL: list_devices(domain="light") to discover available devices
|
||||||
|
3. CHECK: Look for EXACT match `light.<room_name>` first!
|
||||||
|
- For "study lights" → look for `light.study` (not light.study_main, not light.studeerlamp)
|
||||||
|
- For "kitchen lights" → look for `light.kitchen` (not light.kitchen_spot_1)
|
||||||
|
- These room groups control ALL lights in that room at once
|
||||||
|
- If found, use ONLY the group (stop looking for individual lights)
|
||||||
|
4. FALLBACK: Only if no exact room group exists, find entity_ids containing the room name
|
||||||
|
5. CALL: turn_on/turn_off using the EXACT entity_id from step 3 or 4
|
||||||
|
|
||||||
|
Example for "Turn off study lights":
|
||||||
|
1. Domain is "light"
|
||||||
|
2. Call list_devices(domain="light")
|
||||||
|
3. Look for room group: `light.study` - FOUND!
|
||||||
|
4. Call turn_off(entity_id="light.study") # This controls all study lights
|
||||||
|
|
||||||
|
Example for "Turn off hallway lights" (no room group):
|
||||||
|
1. Domain is "light"
|
||||||
|
2. Call list_devices(domain="light")
|
||||||
|
3. Look for room group: `light.hallway` - NOT FOUND
|
||||||
|
4. Find all with "hallway": light.hallway_spot_1, light.hallway_spot_2
|
||||||
|
5. Call turn_off for each
|
||||||
|
|
||||||
|
## Tool Parameter Names
|
||||||
|
|
||||||
|
- turn_on, turn_off, toggle: Use `entity_id` (NOT device_id, NOT id)
|
||||||
|
- activate_scene: Use `scene_id`
|
||||||
|
- run_script: Use `script_id`
|
||||||
|
|
||||||
|
## What NOT To Do
|
||||||
|
|
||||||
|
- NEVER guess an entity_id
|
||||||
|
- NEVER construct an entity_id from the room name
|
||||||
|
- NEVER drop the domain prefix (light., switch., etc.)
|
||||||
|
- NEVER use "device_id" - the parameter is called "entity_id"
|
||||||
|
- NEVER provide an answer without calling list_devices first
|
||||||
|
|
||||||
|
## Response Format
|
||||||
|
|
||||||
|
After completing actions, briefly confirm:
|
||||||
|
- Which devices were affected (list the entity_ids)
|
||||||
|
- Whether each action succeeded or failed
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Lazy initialization to avoid connection issues during imports
|
||||||
|
_housekeeper_agent: Agent[None, str] | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _create_housekeeper_agent() -> Agent[None, str]:
|
||||||
|
"""Create the Housekeeper PydanticAI agent."""
|
||||||
|
from src.anthropic.model_selector import get_model
|
||||||
|
|
||||||
|
# Get best available model (Claude if available, else Ollama)
|
||||||
|
model = get_model()
|
||||||
|
|
||||||
|
agent: Agent[None, str] = Agent(
|
||||||
|
model=model,
|
||||||
|
system_prompt=HOUSEKEEPER_SYSTEM_PROMPT,
|
||||||
|
retries=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Register discovery tools
|
||||||
|
agent.tool_plain(list_areas)
|
||||||
|
agent.tool_plain(list_devices)
|
||||||
|
agent.tool_plain(get_device_state)
|
||||||
|
|
||||||
|
# Register control tools
|
||||||
|
agent.tool_plain(turn_on)
|
||||||
|
agent.tool_plain(turn_off)
|
||||||
|
agent.tool_plain(toggle)
|
||||||
|
|
||||||
|
# Register scene tools
|
||||||
|
agent.tool_plain(list_scenes)
|
||||||
|
agent.tool_plain(activate_scene)
|
||||||
|
|
||||||
|
# Register script tools
|
||||||
|
agent.tool_plain(list_scripts)
|
||||||
|
agent.tool_plain(run_script)
|
||||||
|
|
||||||
|
# Register automation tools
|
||||||
|
agent.tool_plain(list_automations)
|
||||||
|
agent.tool_plain(toggle_automation)
|
||||||
|
|
||||||
|
# Register history tools
|
||||||
|
agent.tool_plain(get_history)
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model_info
|
||||||
|
|
||||||
|
model_info = get_model_info()
|
||||||
|
logger.info(
|
||||||
|
"housekeeper_agent_created",
|
||||||
|
backend=model_info["backend"],
|
||||||
|
model=model_info["model"],
|
||||||
|
tool_count=13,
|
||||||
|
)
|
||||||
|
|
||||||
|
return agent
|
||||||
|
|
||||||
|
|
||||||
|
def get_housekeeper_agent() -> Agent[None, str]:
|
||||||
|
"""
|
||||||
|
Get the Housekeeper agent instance (lazy initialization).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PydanticAI Agent configured for home automation tasks
|
||||||
|
"""
|
||||||
|
global _housekeeper_agent
|
||||||
|
if _housekeeper_agent is None:
|
||||||
|
_housekeeper_agent = _create_housekeeper_agent()
|
||||||
|
return _housekeeper_agent
|
||||||
|
|
||||||
|
|
||||||
|
async def run_housekeeper(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
message_history: list[Any] | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Execute a home automation task with The Housekeeper.
|
||||||
|
|
||||||
|
This is the main entry point for delegating home automation tasks
|
||||||
|
to The Housekeeper from Tatlock or other agents.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: The home automation task or request
|
||||||
|
context: Additional context from conversation
|
||||||
|
message_history: Optional conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Results and confirmation of actions
|
||||||
|
|
||||||
|
Example:
|
||||||
|
result = await run_housekeeper(
|
||||||
|
task="Turn on the living room lights",
|
||||||
|
context="It's evening",
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
agent = get_housekeeper_agent()
|
||||||
|
|
||||||
|
# Build prompt with context if provided
|
||||||
|
prompt = task
|
||||||
|
if context:
|
||||||
|
prompt = f"Context: {context}\n\nTask: {task}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"housekeeper_task_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
has_history=bool(message_history),
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Temperature 0.1 for slight exploration (skipped on Claude backend)
|
||||||
|
from src.anthropic.model_selector import get_sampling_settings
|
||||||
|
|
||||||
|
result = await agent.run(
|
||||||
|
prompt,
|
||||||
|
message_history=message_history,
|
||||||
|
model_settings=get_sampling_settings(0.1),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"housekeeper_task_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(result.output),
|
||||||
|
)
|
||||||
|
|
||||||
|
return result.output
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"housekeeper_task_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
return f"The Housekeeper encountered an error: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def run_housekeeper_stream(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
message_history: list[Any] | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Execute a home automation task with streaming output.
|
||||||
|
|
||||||
|
Yields text deltas as The Housekeeper generates the response.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: The home automation task or request
|
||||||
|
context: Additional context from conversation
|
||||||
|
message_history: Optional conversation history
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
str: Text deltas from the response
|
||||||
|
|
||||||
|
Example:
|
||||||
|
async for delta in run_housekeeper_stream("Turn on the lights"):
|
||||||
|
print(delta, end="", flush=True)
|
||||||
|
"""
|
||||||
|
agent = get_housekeeper_agent()
|
||||||
|
|
||||||
|
# Build prompt with context if provided
|
||||||
|
prompt = task
|
||||||
|
if context:
|
||||||
|
prompt = f"Context: {context}\n\nTask: {task}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"housekeeper_stream_started",
|
||||||
|
task=task[:100],
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Temperature 0.1 for slight exploration (skipped on Claude backend)
|
||||||
|
from src.anthropic.model_selector import get_sampling_settings
|
||||||
|
|
||||||
|
async with agent.run_stream(
|
||||||
|
prompt,
|
||||||
|
message_history=message_history,
|
||||||
|
model_settings=get_sampling_settings(0.1),
|
||||||
|
) as response:
|
||||||
|
async for delta in response.stream_text(delta=True):
|
||||||
|
yield delta
|
||||||
|
|
||||||
|
logger.info("housekeeper_stream_completed", task=task[:50])
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"housekeeper_stream_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
yield f"\n\nThe Housekeeper encountered an error: {str(e)}"
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
"""
|
||||||
|
Housekeeper capability registration for the Household Registry.
|
||||||
|
|
||||||
|
Defines The Housekeeper's capabilities and registers it as a
|
||||||
|
household member for coordination by the Steward and Tatlock.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.housekeeper.agent import get_housekeeper_agent
|
||||||
|
from src.agents.housekeeper.tools import HOUSEKEEPER_TOOLS
|
||||||
|
from src.core.household_registry import (
|
||||||
|
HouseholdCapability,
|
||||||
|
get_household_registry,
|
||||||
|
)
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# The Housekeeper's capability summary for Steward coordination
|
||||||
|
HOUSEKEEPER_CAPABILITY = HouseholdCapability(
|
||||||
|
name="housekeeper",
|
||||||
|
role="The Housekeeper",
|
||||||
|
category="automation",
|
||||||
|
description=(
|
||||||
|
"Home automation control: TURN ON/OFF devices, ACTIVATE scenes, "
|
||||||
|
"RUN scripts, LIST devices, MANAGE automations. Controls lights, "
|
||||||
|
"switches, climate, and other smart home devices via Home Assistant."
|
||||||
|
),
|
||||||
|
domains=[
|
||||||
|
"lights",
|
||||||
|
"switches",
|
||||||
|
"automation",
|
||||||
|
"home",
|
||||||
|
"smart home",
|
||||||
|
"scene",
|
||||||
|
"script",
|
||||||
|
"device",
|
||||||
|
"turn on",
|
||||||
|
"turn off",
|
||||||
|
"temperature",
|
||||||
|
"climate",
|
||||||
|
"fan",
|
||||||
|
"cover",
|
||||||
|
"blinds",
|
||||||
|
],
|
||||||
|
cost="low", # Fast local API calls to core-api
|
||||||
|
requires_network=True, # Needs core-api access
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_housekeeper_capability() -> HouseholdCapability:
|
||||||
|
"""Get The Housekeeper's capability definition."""
|
||||||
|
return HOUSEKEEPER_CAPABILITY
|
||||||
|
|
||||||
|
|
||||||
|
def register_housekeeper() -> None:
|
||||||
|
"""
|
||||||
|
Register The Housekeeper with the Household Registry.
|
||||||
|
|
||||||
|
This makes The Housekeeper available for:
|
||||||
|
- Steward recommendations (via capability summary)
|
||||||
|
- Tatlock delegation (via agent reference)
|
||||||
|
- Tool scoping (via tool list)
|
||||||
|
"""
|
||||||
|
registry = get_household_registry()
|
||||||
|
|
||||||
|
# Check if already registered
|
||||||
|
if "housekeeper" in registry:
|
||||||
|
logger.debug("housekeeper_already_registered")
|
||||||
|
return
|
||||||
|
|
||||||
|
registry.register(
|
||||||
|
name="housekeeper",
|
||||||
|
capability=HOUSEKEEPER_CAPABILITY,
|
||||||
|
tools=HOUSEKEEPER_TOOLS,
|
||||||
|
agent=get_housekeeper_agent(),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"housekeeper_registered",
|
||||||
|
role=HOUSEKEEPER_CAPABILITY.role,
|
||||||
|
domains=HOUSEKEEPER_CAPABILITY.domains,
|
||||||
|
tool_count=len(HOUSEKEEPER_TOOLS),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def unregister_housekeeper() -> None:
|
||||||
|
"""Unregister The Housekeeper from the Household Registry."""
|
||||||
|
registry = get_household_registry()
|
||||||
|
registry.unregister("housekeeper")
|
||||||
|
logger.info("housekeeper_unregistered")
|
||||||
@@ -0,0 +1,556 @@
|
|||||||
|
"""
|
||||||
|
HTTP client for the Core-API service.
|
||||||
|
|
||||||
|
Provides async methods for home automation operations via Home Assistant.
|
||||||
|
Core-API is a separate service that wraps the Home Assistant REST API
|
||||||
|
into LLM-friendly endpoints.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
|
from src.core.config import config
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Response Models
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
class Device(BaseModel):
|
||||||
|
"""Device from Home Assistant."""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
name: str
|
||||||
|
state: str
|
||||||
|
domain: str
|
||||||
|
area: str | None = None
|
||||||
|
attributes: dict[str, Any] = Field(default_factory=dict)
|
||||||
|
|
||||||
|
|
||||||
|
class DeviceState(BaseModel):
|
||||||
|
"""Detailed state of a device."""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
state: str
|
||||||
|
attributes: dict[str, Any] = Field(default_factory=dict)
|
||||||
|
last_changed: str | None = None
|
||||||
|
last_updated: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class Scene(BaseModel):
|
||||||
|
"""Scene from Home Assistant."""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
name: str
|
||||||
|
friendly_name: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class Script(BaseModel):
|
||||||
|
"""Script from Home Assistant."""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
name: str
|
||||||
|
description: str | None = None
|
||||||
|
last_triggered: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class Automation(BaseModel):
|
||||||
|
"""Automation from Home Assistant."""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
name: str
|
||||||
|
state: str = "on"
|
||||||
|
last_triggered: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class HistoryEntry(BaseModel):
|
||||||
|
"""History entry for an entity."""
|
||||||
|
|
||||||
|
state: str
|
||||||
|
timestamp: str
|
||||||
|
attributes: dict[str, Any] = Field(default_factory=dict)
|
||||||
|
|
||||||
|
|
||||||
|
class ControlResult(BaseModel):
|
||||||
|
"""Result of a device control operation."""
|
||||||
|
|
||||||
|
success: bool
|
||||||
|
entity_id: str
|
||||||
|
action: str
|
||||||
|
message: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
class Area(BaseModel):
|
||||||
|
"""Area/room from Home Assistant."""
|
||||||
|
|
||||||
|
area_id: str
|
||||||
|
name: str
|
||||||
|
device_count: int = 0
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Client
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
class CoreAPIClient:
|
||||||
|
"""
|
||||||
|
Async HTTP client for Core-API (Home Assistant wrapper).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
devices = await client.list_devices()
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
base_url: str | None = None,
|
||||||
|
api_key: str | None = None,
|
||||||
|
timeout: int = 30,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize the client.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
base_url: Core-API URL (defaults to config)
|
||||||
|
api_key: API key for authentication (defaults to config)
|
||||||
|
timeout: Request timeout in seconds
|
||||||
|
"""
|
||||||
|
self.base_url = base_url or str(config.CORE_API_HOST)
|
||||||
|
self.api_key = api_key or config.CORE_API_KEY
|
||||||
|
self.timeout = timeout
|
||||||
|
self._client: httpx.AsyncClient | None = None
|
||||||
|
|
||||||
|
async def __aenter__(self) -> "CoreAPIClient":
|
||||||
|
"""Create HTTP client on context entry."""
|
||||||
|
headers = {}
|
||||||
|
if self.api_key:
|
||||||
|
headers["Authorization"] = f"Bearer {self.api_key}"
|
||||||
|
|
||||||
|
self._client = httpx.AsyncClient(
|
||||||
|
base_url=self.base_url,
|
||||||
|
headers=headers,
|
||||||
|
timeout=self.timeout,
|
||||||
|
)
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
|
||||||
|
"""Close HTTP client on context exit."""
|
||||||
|
if self._client:
|
||||||
|
await self._client.aclose()
|
||||||
|
self._client = None
|
||||||
|
|
||||||
|
def _ensure_client(self) -> httpx.AsyncClient:
|
||||||
|
"""Ensure client is initialized."""
|
||||||
|
if self._client is None:
|
||||||
|
raise RuntimeError(
|
||||||
|
"Client not initialized. Use 'async with CoreAPIClient() as client:'"
|
||||||
|
)
|
||||||
|
return self._client
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Device Discovery
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def list_devices(
|
||||||
|
self,
|
||||||
|
domain: str | None = None,
|
||||||
|
area: str | None = None,
|
||||||
|
) -> list[Device]:
|
||||||
|
"""
|
||||||
|
List devices, optionally filtered by domain or area.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
domain: Filter by domain (light, switch, climate, etc.)
|
||||||
|
area: Filter by area (living_room, bedroom, etc.)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of devices matching filters
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
params: dict[str, str] = {}
|
||||||
|
if domain:
|
||||||
|
params["domain"] = domain
|
||||||
|
if area:
|
||||||
|
params["area"] = area
|
||||||
|
|
||||||
|
logger.debug("core_api_list_devices", domain=domain, area=area)
|
||||||
|
|
||||||
|
response = await client.get("/housekeeping/devices", params=params or None)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [Device(**d) for d in data.get("devices", [])]
|
||||||
|
|
||||||
|
async def list_areas(self) -> list[Area]:
|
||||||
|
"""
|
||||||
|
List all areas/rooms in Home Assistant.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of areas with device counts
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_list_areas")
|
||||||
|
|
||||||
|
response = await client.get("/housekeeping/areas")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [Area(**a) for a in data.get("areas", [])]
|
||||||
|
|
||||||
|
async def get_device_state(self, entity_id: str) -> DeviceState:
|
||||||
|
"""
|
||||||
|
Get the current state of a specific device.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Home Assistant entity ID (e.g., light.living_room)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Current device state with attributes
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_get_state", entity_id=entity_id)
|
||||||
|
|
||||||
|
response = await client.get(f"/housekeeping/devices/{entity_id}")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
return DeviceState(**response.json())
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Device Control
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def turn_on(
|
||||||
|
self,
|
||||||
|
entity_id: str,
|
||||||
|
brightness: int | None = None,
|
||||||
|
color_temp: int | None = None,
|
||||||
|
rgb_color: tuple[int, int, int] | None = None,
|
||||||
|
) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Turn on a device.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Device to turn on
|
||||||
|
brightness: Optional brightness (0-255) for lights
|
||||||
|
color_temp: Optional color temperature in Kelvin for lights
|
||||||
|
rgb_color: Optional RGB color tuple for lights
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
payload: dict[str, Any] = {"action": "turn_on"}
|
||||||
|
if brightness is not None:
|
||||||
|
payload["brightness"] = brightness
|
||||||
|
if color_temp is not None:
|
||||||
|
payload["color_temp"] = color_temp
|
||||||
|
if rgb_color is not None:
|
||||||
|
payload["rgb_color"] = list(rgb_color)
|
||||||
|
|
||||||
|
logger.info("core_api_turn_on", entity_id=entity_id, payload=payload)
|
||||||
|
|
||||||
|
response = await client.post(
|
||||||
|
f"/housekeeping/devices/{entity_id}/control",
|
||||||
|
json=payload,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=entity_id,
|
||||||
|
action="turn_on",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def turn_off(self, entity_id: str) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Turn off a device.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Device to turn off
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.info("core_api_turn_off", entity_id=entity_id)
|
||||||
|
|
||||||
|
response = await client.post(
|
||||||
|
f"/housekeeping/devices/{entity_id}/control",
|
||||||
|
json={"action": "turn_off"},
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=entity_id,
|
||||||
|
action="turn_off",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def toggle(self, entity_id: str) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Toggle a device's state.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Device to toggle
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.info("core_api_toggle", entity_id=entity_id)
|
||||||
|
|
||||||
|
response = await client.post(
|
||||||
|
f"/housekeeping/devices/{entity_id}/control",
|
||||||
|
json={"action": "toggle"},
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=entity_id,
|
||||||
|
action="toggle",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Scenes
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def list_scenes(self) -> list[Scene]:
|
||||||
|
"""
|
||||||
|
List all available scenes.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of scenes
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_list_scenes")
|
||||||
|
|
||||||
|
response = await client.get("/housekeeping/scenes")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [Scene(**s) for s in data.get("scenes", [])]
|
||||||
|
|
||||||
|
async def activate_scene(self, scene_id: str) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Activate a scene.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
scene_id: Scene entity ID (e.g., scene.movie_night)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.info("core_api_activate_scene", scene_id=scene_id)
|
||||||
|
|
||||||
|
response = await client.post(f"/housekeeping/scenes/{scene_id}/activate")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=scene_id,
|
||||||
|
action="activate",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Scripts
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def list_scripts(self) -> list[Script]:
|
||||||
|
"""
|
||||||
|
List all available scripts.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of scripts
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_list_scripts")
|
||||||
|
|
||||||
|
response = await client.get("/housekeeping/scripts")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [Script(**s) for s in data.get("scripts", [])]
|
||||||
|
|
||||||
|
async def run_script(
|
||||||
|
self,
|
||||||
|
script_id: str,
|
||||||
|
variables: dict[str, Any] | None = None,
|
||||||
|
) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Run a script.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
script_id: Script entity ID (e.g., script.good_morning)
|
||||||
|
variables: Optional variables to pass to the script
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
payload: dict[str, Any] = {}
|
||||||
|
if variables:
|
||||||
|
payload["variables"] = variables
|
||||||
|
|
||||||
|
logger.info("core_api_run_script", script_id=script_id)
|
||||||
|
|
||||||
|
response = await client.post(
|
||||||
|
f"/housekeeping/scripts/{script_id}/run",
|
||||||
|
json=payload or None,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=script_id,
|
||||||
|
action="run",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Automations
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def list_automations(self) -> list[Automation]:
|
||||||
|
"""
|
||||||
|
List all automations.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of automations with their states
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_list_automations")
|
||||||
|
|
||||||
|
response = await client.get("/housekeeping/automations")
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [Automation(**a) for a in data.get("automations", [])]
|
||||||
|
|
||||||
|
async def toggle_automation(
|
||||||
|
self,
|
||||||
|
automation_id: str,
|
||||||
|
enable: bool,
|
||||||
|
) -> ControlResult:
|
||||||
|
"""
|
||||||
|
Enable or disable an automation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
automation_id: Automation entity ID
|
||||||
|
enable: True to enable, False to disable
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Result of the operation
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"core_api_toggle_automation",
|
||||||
|
automation_id=automation_id,
|
||||||
|
enable=enable,
|
||||||
|
)
|
||||||
|
|
||||||
|
response = await client.post(
|
||||||
|
f"/housekeeping/automations/{automation_id}/toggle",
|
||||||
|
json={"enable": enable},
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return ControlResult(
|
||||||
|
success=data.get("success", True),
|
||||||
|
entity_id=automation_id,
|
||||||
|
action="enable" if enable else "disable",
|
||||||
|
message=data.get("message", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# History
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def get_history(
|
||||||
|
self,
|
||||||
|
entity_id: str,
|
||||||
|
hours: int = 24,
|
||||||
|
) -> list[HistoryEntry]:
|
||||||
|
"""
|
||||||
|
Get history for an entity.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Entity to get history for
|
||||||
|
hours: Number of hours of history (default: 24)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of historical state entries
|
||||||
|
"""
|
||||||
|
client = self._ensure_client()
|
||||||
|
|
||||||
|
logger.debug("core_api_get_history", entity_id=entity_id, hours=hours)
|
||||||
|
|
||||||
|
response = await client.get(
|
||||||
|
"/housekeeping/history",
|
||||||
|
params={"entity_id": entity_id, "hours": hours},
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
return [HistoryEntry(**h) for h in data.get("history", [])]
|
||||||
|
|
||||||
|
# ========================================================================
|
||||||
|
# Health Check
|
||||||
|
# ========================================================================
|
||||||
|
|
||||||
|
async def health_check(self) -> bool:
|
||||||
|
"""
|
||||||
|
Check if core-api and Home Assistant are healthy.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if healthy, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = self._ensure_client()
|
||||||
|
response = await client.get("/housekeeping/health")
|
||||||
|
return response.status_code == 200
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("core_api_health_check_failed", error=str(e))
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# Global client factory
|
||||||
|
async def get_core_api_client() -> CoreAPIClient:
|
||||||
|
"""
|
||||||
|
Get a core-api client instance.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
async with get_core_api_client() as client:
|
||||||
|
devices = await client.list_devices()
|
||||||
|
"""
|
||||||
|
return CoreAPIClient()
|
||||||
@@ -0,0 +1,592 @@
|
|||||||
|
"""
|
||||||
|
Housekeeper tools for PydanticAI agent.
|
||||||
|
|
||||||
|
These tools wrap the core-api service and are registered with
|
||||||
|
The Housekeeper agent for home automation tasks.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.housekeeper.client import CoreAPIClient
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Device Discovery
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def list_devices(
|
||||||
|
domain: str | None = None,
|
||||||
|
area: str | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
List available devices in the smart home.
|
||||||
|
|
||||||
|
Use this to discover what devices can be controlled.
|
||||||
|
Can filter by domain (device type) or area (room).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
domain: Device type filter (light, switch, climate, cover, fan, etc.)
|
||||||
|
area: Room/area filter (living_room, bedroom, kitchen, etc.)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of devices with their current states
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_devices() # All devices
|
||||||
|
list_devices(domain="light") # Only lights
|
||||||
|
list_devices(area="living_room") # Living room devices
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
devices = await client.list_devices(domain=domain, area=area)
|
||||||
|
|
||||||
|
if not devices:
|
||||||
|
filters = []
|
||||||
|
if domain:
|
||||||
|
filters.append(f"domain={domain}")
|
||||||
|
if area:
|
||||||
|
filters.append(f"area={area}")
|
||||||
|
filter_str = f" with filters: {', '.join(filters)}" if filters else ""
|
||||||
|
return f"No devices found{filter_str}"
|
||||||
|
|
||||||
|
# Group by domain for readability
|
||||||
|
by_domain: dict[str, list] = {}
|
||||||
|
for device in devices:
|
||||||
|
by_domain.setdefault(device.domain, []).append(device)
|
||||||
|
|
||||||
|
output_parts = ["## Smart Home Devices\n"]
|
||||||
|
|
||||||
|
for dom, dom_devices in sorted(by_domain.items()):
|
||||||
|
output_parts.append(f"### {dom.title()}s")
|
||||||
|
|
||||||
|
# Sort devices: room groups first (using Home Assistant's is_hue_group attribute)
|
||||||
|
def is_room_group(d: object) -> bool:
|
||||||
|
"""Check if device is a room group based on HA attributes."""
|
||||||
|
attrs = getattr(d, "attributes", {})
|
||||||
|
# Check for Hue room groups
|
||||||
|
if attrs.get("is_hue_group") and attrs.get("hue_type") == "room":
|
||||||
|
return True
|
||||||
|
# Check for other group indicators (icon or entity_id list)
|
||||||
|
if "entity_id" in attrs and isinstance(attrs["entity_id"], list):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
sorted_devices = sorted(
|
||||||
|
dom_devices, key=lambda d: (not is_room_group(d), d.entity_id)
|
||||||
|
)
|
||||||
|
|
||||||
|
for device in sorted_devices:
|
||||||
|
state_icon = (
|
||||||
|
"on"
|
||||||
|
if device.state == "on"
|
||||||
|
else "off"
|
||||||
|
if device.state == "off"
|
||||||
|
else device.state
|
||||||
|
)
|
||||||
|
area_str = f" ({device.area})" if device.area else ""
|
||||||
|
# Mark room groups clearly using actual HA data
|
||||||
|
group_marker = " [ROOM GROUP]" if is_room_group(device) else ""
|
||||||
|
output_parts.append(
|
||||||
|
f"- **{device.name}**{area_str}{group_marker}: {state_icon}"
|
||||||
|
)
|
||||||
|
output_parts.append(f" ID: `{device.entity_id}`")
|
||||||
|
output_parts.append("")
|
||||||
|
|
||||||
|
logger.info("housekeeper_list_devices", count=len(devices))
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_list_devices_error", error=str(e))
|
||||||
|
return f"Error listing devices: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def list_areas() -> str:
|
||||||
|
"""
|
||||||
|
List all areas/rooms in the smart home.
|
||||||
|
|
||||||
|
Use this to discover what rooms/areas are configured in Home Assistant.
|
||||||
|
Useful before filtering devices by area.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of areas with device counts
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_areas() # See all rooms/areas
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
areas = await client.list_areas()
|
||||||
|
|
||||||
|
if not areas:
|
||||||
|
return "No areas found in Home Assistant"
|
||||||
|
|
||||||
|
output_parts = ["## Smart Home Areas\n"]
|
||||||
|
|
||||||
|
for area in sorted(areas, key=lambda a: a.name):
|
||||||
|
device_str = f" ({area.device_count} devices)" if area.device_count else ""
|
||||||
|
output_parts.append(f"- **{area.name}**{device_str}")
|
||||||
|
output_parts.append(f" ID: `{area.area_id}`")
|
||||||
|
|
||||||
|
output_parts.append("")
|
||||||
|
output_parts.append(f"*{len(areas)} areas total*")
|
||||||
|
|
||||||
|
logger.info("housekeeper_list_areas", count=len(areas))
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_list_areas_error", error=str(e))
|
||||||
|
return f"Error listing areas: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def get_device_state(entity_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Get the current state and attributes of a specific device.
|
||||||
|
|
||||||
|
Use this to check a device's detailed status before or after control.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: The device entity ID (e.g., light.living_room, switch.coffee_maker)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Detailed device state including all attributes
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
get_device_state("light.living_room")
|
||||||
|
get_device_state("climate.bedroom")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
state = await client.get_device_state(entity_id)
|
||||||
|
|
||||||
|
output_parts = [
|
||||||
|
f"## Device: {entity_id}",
|
||||||
|
f"**State:** {state.state}",
|
||||||
|
]
|
||||||
|
|
||||||
|
if state.last_changed:
|
||||||
|
output_parts.append(f"**Last Changed:** {state.last_changed}")
|
||||||
|
|
||||||
|
if state.attributes:
|
||||||
|
output_parts.append("\n**Attributes:**")
|
||||||
|
for key, value in state.attributes.items():
|
||||||
|
if key not in ("friendly_name", "entity_id"):
|
||||||
|
output_parts.append(f"- {key}: {value}")
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_get_state_error", error=str(e), entity_id=entity_id)
|
||||||
|
return f"Error getting state for {entity_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Device Control
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def turn_on(
|
||||||
|
entity_id: str,
|
||||||
|
brightness: int | None = None,
|
||||||
|
color_temp: int | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Turn on a device. Use the entity_id parameter with the EXACT value from list_devices.
|
||||||
|
|
||||||
|
For lights, can optionally set brightness and color temperature.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: The EXACT entity ID from list_devices including domain prefix.
|
||||||
|
brightness: Optional brightness for lights (0-255, where 255 is full brightness)
|
||||||
|
color_temp: Optional color temperature in Kelvin (2700=warm, 6500=cool)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of the action
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
turn_on(entity_id="light.living_room")
|
||||||
|
turn_on(entity_id="light.bedroom", brightness=128)
|
||||||
|
turn_on(entity_id="switch.coffee_maker")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.turn_on(
|
||||||
|
entity_id=entity_id,
|
||||||
|
brightness=brightness,
|
||||||
|
color_temp=color_temp,
|
||||||
|
)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
extras = []
|
||||||
|
if brightness is not None:
|
||||||
|
extras.append(f"brightness {brightness}/255")
|
||||||
|
if color_temp is not None:
|
||||||
|
extras.append(f"color temp {color_temp}K")
|
||||||
|
|
||||||
|
extra_str = f" ({', '.join(extras)})" if extras else ""
|
||||||
|
return f"Turned on {entity_id}{extra_str}"
|
||||||
|
else:
|
||||||
|
return f"Failed to turn on {entity_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_turn_on_error", error=str(e), entity_id=entity_id)
|
||||||
|
return f"Error turning on {entity_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def turn_off(entity_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Turn off a device. Use the entity_id parameter with the EXACT value from list_devices.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: The EXACT entity ID from list_devices including domain prefix.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of the action
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
turn_off(entity_id="light.living_room")
|
||||||
|
turn_off(entity_id="switch.coffee_maker")
|
||||||
|
turn_off(entity_id="light.kitchen")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.turn_off(entity_id=entity_id)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
return f"Turned off {entity_id}"
|
||||||
|
else:
|
||||||
|
return f"Failed to turn off {entity_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_turn_off_error", error=str(e), entity_id=entity_id)
|
||||||
|
return f"Error turning off {entity_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def toggle(entity_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Toggle a device's state (on becomes off, off becomes on).
|
||||||
|
|
||||||
|
Use the entity_id parameter with the EXACT value from list_devices.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: The EXACT entity ID from list_devices including domain prefix.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation with the new state
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
toggle(entity_id="light.living_room")
|
||||||
|
toggle(entity_id="switch.fan")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.toggle(entity_id=entity_id)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
return f"Toggled {entity_id}"
|
||||||
|
else:
|
||||||
|
return f"Failed to toggle {entity_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_toggle_error", error=str(e), entity_id=entity_id)
|
||||||
|
return f"Error toggling {entity_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Scenes
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def list_scenes() -> str:
|
||||||
|
"""
|
||||||
|
List all available scenes.
|
||||||
|
|
||||||
|
Scenes are pre-configured combinations of device states.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of available scenes
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_scenes()
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
scenes = await client.list_scenes()
|
||||||
|
|
||||||
|
if not scenes:
|
||||||
|
return "No scenes found"
|
||||||
|
|
||||||
|
output_parts = ["## Available Scenes\n"]
|
||||||
|
for scene in scenes:
|
||||||
|
name = scene.friendly_name or scene.name
|
||||||
|
output_parts.append(f"- **{name}**")
|
||||||
|
output_parts.append(f" ID: `{scene.entity_id}`")
|
||||||
|
|
||||||
|
logger.info("housekeeper_list_scenes", count=len(scenes))
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_list_scenes_error", error=str(e))
|
||||||
|
return f"Error listing scenes: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def activate_scene(scene_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Activate a scene.
|
||||||
|
|
||||||
|
This sets all devices in the scene to their configured states.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
scene_id: Scene entity ID (e.g., scene.movie_night, scene.good_morning)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of activation
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
activate_scene("scene.movie_night")
|
||||||
|
activate_scene("scene.good_morning")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.activate_scene(scene_id=scene_id)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
return f"Activated scene: {scene_id}"
|
||||||
|
else:
|
||||||
|
return f"Failed to activate {scene_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_activate_scene_error", error=str(e), scene_id=scene_id)
|
||||||
|
return f"Error activating scene {scene_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Scripts
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def list_scripts() -> str:
|
||||||
|
"""
|
||||||
|
List all available automation scripts.
|
||||||
|
|
||||||
|
Scripts are sequences of actions that can be triggered manually.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of available scripts
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_scripts()
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
scripts = await client.list_scripts()
|
||||||
|
|
||||||
|
if not scripts:
|
||||||
|
return "No scripts found"
|
||||||
|
|
||||||
|
output_parts = ["## Available Scripts\n"]
|
||||||
|
for script in scripts:
|
||||||
|
output_parts.append(f"- **{script.name}**")
|
||||||
|
if script.description:
|
||||||
|
output_parts.append(f" {script.description}")
|
||||||
|
output_parts.append(f" ID: `{script.entity_id}`")
|
||||||
|
if script.last_triggered:
|
||||||
|
output_parts.append(f" Last run: {script.last_triggered}")
|
||||||
|
|
||||||
|
logger.info("housekeeper_list_scripts", count=len(scripts))
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_list_scripts_error", error=str(e))
|
||||||
|
return f"Error listing scripts: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def run_script(script_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Run an automation script.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
script_id: Script entity ID (e.g., script.good_morning, script.bedtime)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of execution
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
run_script("script.good_morning")
|
||||||
|
run_script("script.all_lights_off")
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.run_script(script_id=script_id)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
return f"Running script: {script_id}"
|
||||||
|
else:
|
||||||
|
return f"Failed to run {script_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_run_script_error", error=str(e), script_id=script_id)
|
||||||
|
return f"Error running script {script_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Automations
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def list_automations() -> str:
|
||||||
|
"""
|
||||||
|
List all automations and their current states.
|
||||||
|
|
||||||
|
Automations are event-triggered rules that run automatically.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of automations with enabled/disabled status
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
list_automations()
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
automations = await client.list_automations()
|
||||||
|
|
||||||
|
if not automations:
|
||||||
|
return "No automations found"
|
||||||
|
|
||||||
|
output_parts = ["## Automations\n"]
|
||||||
|
|
||||||
|
# Group by state
|
||||||
|
enabled = [a for a in automations if a.state == "on"]
|
||||||
|
disabled = [a for a in automations if a.state != "on"]
|
||||||
|
|
||||||
|
if enabled:
|
||||||
|
output_parts.append("### Enabled")
|
||||||
|
for auto in enabled:
|
||||||
|
output_parts.append(f"- **{auto.name}**")
|
||||||
|
output_parts.append(f" ID: `{auto.entity_id}`")
|
||||||
|
if auto.last_triggered:
|
||||||
|
output_parts.append(f" Last triggered: {auto.last_triggered}")
|
||||||
|
output_parts.append("")
|
||||||
|
|
||||||
|
if disabled:
|
||||||
|
output_parts.append("### Disabled")
|
||||||
|
for auto in disabled:
|
||||||
|
output_parts.append(f"- **{auto.name}**")
|
||||||
|
output_parts.append(f" ID: `{auto.entity_id}`")
|
||||||
|
|
||||||
|
logger.info("housekeeper_list_automations", count=len(automations))
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_list_automations_error", error=str(e))
|
||||||
|
return f"Error listing automations: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
async def toggle_automation(automation_id: str, enable: bool) -> str:
|
||||||
|
"""
|
||||||
|
Enable or disable an automation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
automation_id: Automation entity ID
|
||||||
|
enable: True to enable, False to disable
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Confirmation of the change
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
toggle_automation("automation.morning_lights", enable=True)
|
||||||
|
toggle_automation("automation.vacation_mode", enable=False)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
result = await client.toggle_automation(
|
||||||
|
automation_id=automation_id,
|
||||||
|
enable=enable,
|
||||||
|
)
|
||||||
|
|
||||||
|
action = "Enabled" if enable else "Disabled"
|
||||||
|
if result.success:
|
||||||
|
return f"{action} automation: {automation_id}"
|
||||||
|
else:
|
||||||
|
return f"Failed to {action.lower()} {automation_id}: {result.message}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"housekeeper_toggle_automation_error",
|
||||||
|
error=str(e),
|
||||||
|
automation_id=automation_id,
|
||||||
|
)
|
||||||
|
return f"Error toggling automation {automation_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# History
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
async def get_history(entity_id: str, hours: int = 24) -> str:
|
||||||
|
"""
|
||||||
|
Get the state history of a device.
|
||||||
|
|
||||||
|
Useful for understanding patterns or troubleshooting.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Device to get history for
|
||||||
|
hours: Number of hours of history (default: 24)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of state changes over the time period
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
get_history("light.living_room")
|
||||||
|
get_history("climate.bedroom", hours=48)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with CoreAPIClient() as client:
|
||||||
|
history = await client.get_history(entity_id=entity_id, hours=hours)
|
||||||
|
|
||||||
|
if not history:
|
||||||
|
return f"No history found for {entity_id} in the last {hours} hours"
|
||||||
|
|
||||||
|
output_parts = [f"## History: {entity_id}", f"*Last {hours} hours*\n"]
|
||||||
|
|
||||||
|
for entry in history[-20:]: # Show last 20 entries
|
||||||
|
output_parts.append(f"- **{entry.timestamp}**: {entry.state}")
|
||||||
|
|
||||||
|
if len(history) > 20:
|
||||||
|
output_parts.append(f"\n*(showing last 20 of {len(history)} entries)*")
|
||||||
|
|
||||||
|
return "\n".join(output_parts)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("housekeeper_get_history_error", error=str(e), entity_id=entity_id)
|
||||||
|
return f"Error getting history for {entity_id}: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Tool Collection for Registration
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
# All tools available to The Housekeeper
|
||||||
|
HOUSEKEEPER_TOOLS = [
|
||||||
|
# Discovery
|
||||||
|
list_areas,
|
||||||
|
list_devices,
|
||||||
|
get_device_state,
|
||||||
|
# Control
|
||||||
|
turn_on,
|
||||||
|
turn_off,
|
||||||
|
toggle,
|
||||||
|
# Scenes
|
||||||
|
list_scenes,
|
||||||
|
activate_scene,
|
||||||
|
# Scripts
|
||||||
|
list_scripts,
|
||||||
|
run_script,
|
||||||
|
# Automations
|
||||||
|
list_automations,
|
||||||
|
toggle_automation,
|
||||||
|
# History
|
||||||
|
get_history,
|
||||||
|
]
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
"""
|
||||||
|
The Librarian - Expert agent for research and knowledge management.
|
||||||
|
|
||||||
|
Connects to the library-desk API to provide:
|
||||||
|
- HybridRAG search (vector + graph + web)
|
||||||
|
- Wiki.js operations
|
||||||
|
- Knowledge graph queries
|
||||||
|
- Semantic search
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.librarian.agent import (
|
||||||
|
get_librarian_agent,
|
||||||
|
run_librarian,
|
||||||
|
)
|
||||||
|
from src.agents.librarian.capability import (
|
||||||
|
LIBRARIAN_CAPABILITY,
|
||||||
|
get_librarian_capability,
|
||||||
|
register_librarian,
|
||||||
|
unregister_librarian,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"LIBRARIAN_CAPABILITY",
|
||||||
|
"get_librarian_capability",
|
||||||
|
"get_librarian_agent",
|
||||||
|
"register_librarian",
|
||||||
|
"unregister_librarian",
|
||||||
|
"run_librarian",
|
||||||
|
]
|
||||||
@@ -0,0 +1,299 @@
|
|||||||
|
"""
|
||||||
|
The Librarian - Expert agent for research and knowledge management.
|
||||||
|
|
||||||
|
A PydanticAI agent that provides research assistance through
|
||||||
|
the library-desk API, offering:
|
||||||
|
- HybridRAG search across all knowledge sources
|
||||||
|
- Wiki and document management
|
||||||
|
- Semantic search and knowledge graph exploration
|
||||||
|
"""
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from pydantic_ai import Agent
|
||||||
|
|
||||||
|
from src.agents.librarian.client import library_client_session
|
||||||
|
from src.agents.librarian.tools import (
|
||||||
|
create_wiki_page,
|
||||||
|
explore_knowledge_graph,
|
||||||
|
find_related_entities,
|
||||||
|
get_dossier_pages,
|
||||||
|
get_wiki_page,
|
||||||
|
hybrid_search,
|
||||||
|
list_dossiers,
|
||||||
|
read_url,
|
||||||
|
read_urls_batch,
|
||||||
|
search_web,
|
||||||
|
search_wiki,
|
||||||
|
semantic_search,
|
||||||
|
smart_create_wiki_page,
|
||||||
|
update_wiki_page,
|
||||||
|
)
|
||||||
|
from src.agents.protocol import AgentError
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
# Librarian system prompt
|
||||||
|
LIBRARIAN_SYSTEM_PROMPT = """You are The Librarian, an expert research assistant in the Tatlock household.
|
||||||
|
|
||||||
|
Your role is to help users find, understand, synthesize, and manage information from:
|
||||||
|
- The personal wiki (Wiki.js) containing documentation and notes
|
||||||
|
- The knowledge graph (Neo4j) with entities and relationships
|
||||||
|
- Vector embeddings (Qdrant) for semantic search
|
||||||
|
- Paperless documents (📑) - indexed PDFs, scanned documents, invoices, receipts from the user's document archive
|
||||||
|
- Volatile cache (⚡) - pre-fetched real-time data for user-relevant locations and items:
|
||||||
|
- weather/forecast: conditions and forecasts for user's configured cities
|
||||||
|
- news: headlines from user's preferred sources
|
||||||
|
- stock/crypto: quotes for user's watched symbols
|
||||||
|
- sun/air_quality: data for user's locations
|
||||||
|
- Note: volatile data may not exist for arbitrary queries - falls back to web search
|
||||||
|
- Web search (SearXNG) for current information not available in cache
|
||||||
|
|
||||||
|
## Your Personality
|
||||||
|
- Scholarly and thorough in your research
|
||||||
|
- Cite your sources and provide context
|
||||||
|
- Organize information clearly
|
||||||
|
- Suggest related topics when relevant
|
||||||
|
- Acknowledge limitations when information is incomplete
|
||||||
|
|
||||||
|
## Your Tools
|
||||||
|
|
||||||
|
### Web Search & Content Extraction
|
||||||
|
- **search_web**: Search the internet for current information (weather, news, facts)
|
||||||
|
- Use for: weather forecasts, current events, recent developments, external facts
|
||||||
|
- Returns extracted content from search results, not just snippets
|
||||||
|
- **read_url**: Read and extract content from a specific URL
|
||||||
|
- Use when: user provides a URL or you need to read a specific webpage
|
||||||
|
- **read_urls_batch**: Read multiple URLs in parallel (up to 20)
|
||||||
|
- Use for: comparing multiple sources, gathering info from several pages
|
||||||
|
|
||||||
|
### Internal Research Tools
|
||||||
|
- **hybrid_search**: Your primary research tool - searches ALL sources at once:
|
||||||
|
- Wiki pages (vector similarity)
|
||||||
|
- Knowledge graph (entity relationships)
|
||||||
|
- Paperless documents (📑 indexed PDFs, scans)
|
||||||
|
- Volatile cache (⚡ weather, news, stocks - when available)
|
||||||
|
- Web search (current information)
|
||||||
|
Results are fused and re-ranked by relevance. Volatile data gets priority when fresh.
|
||||||
|
- **search_wiki**: Find specific wiki pages by keyword
|
||||||
|
- **semantic_search**: Find conceptually similar content
|
||||||
|
- **explore_knowledge_graph** / **find_related_entities**: Discover connections
|
||||||
|
- **list_dossiers** / **get_dossier_pages**: Browse knowledge collections
|
||||||
|
|
||||||
|
### Wiki Reading Tools
|
||||||
|
- **get_wiki_page**: Read full content of a wiki page by ID
|
||||||
|
- ALWAYS use this to fetch and read page content when summarizing
|
||||||
|
- Use after search_wiki to get the full text of a specific page
|
||||||
|
|
||||||
|
### Wiki Writing Tools
|
||||||
|
- **smart_create_wiki_page**: Create a page with automatic research (PREFERRED)
|
||||||
|
- **This is the DEFAULT choice when user asks to create a wiki page about a topic**
|
||||||
|
- When user says "Create a page about X" or "Add X to the wiki" without providing specific content, ALWAYS use this tool
|
||||||
|
- Automatically researches the topic from wiki, graph, and web
|
||||||
|
- Synthesizes content with proper source attribution
|
||||||
|
- Creates bidirectional links in knowledge graph
|
||||||
|
- **create_wiki_page**: Create a page with user-provided content
|
||||||
|
- ONLY use when user provides specific text/content they want added verbatim
|
||||||
|
- For simple notes, reminders, or quick additions with exact content
|
||||||
|
- **update_wiki_page**: Update an existing page (partial updates)
|
||||||
|
- Use when: "Update the page about X", "Fix this info", "Add to dossier"
|
||||||
|
- First search_wiki to find the page, then get_wiki_page to read it
|
||||||
|
- Only specify fields you want to change
|
||||||
|
|
||||||
|
## Research Approach
|
||||||
|
1. Start with hybrid_search for broad queries
|
||||||
|
2. Use search_wiki for specific document lookups
|
||||||
|
3. **ALWAYS use get_wiki_page to fetch full content** before summarizing a page
|
||||||
|
4. Use semantic_search when looking for conceptually similar content
|
||||||
|
5. Explore the knowledge graph to find connections between concepts
|
||||||
|
6. Synthesize and summarize findings clearly
|
||||||
|
|
||||||
|
## Writing Approach
|
||||||
|
When asked to create or update wiki content:
|
||||||
|
1. **"Create a page about X" (no specific content provided)**: Use smart_create_wiki_page
|
||||||
|
- This is the PREFERRED tool for topic-based page creation
|
||||||
|
- It researches first and creates comprehensive, well-sourced content
|
||||||
|
2. **User provides exact text to add**: Use create_wiki_page with their content
|
||||||
|
3. **Updating existing pages**:
|
||||||
|
- Search for the page with search_wiki
|
||||||
|
- Fetch full content with get_wiki_page
|
||||||
|
- Make edits and use update_wiki_page
|
||||||
|
4. **Organizing into dossiers**: Use update_wiki_page with just the tags field
|
||||||
|
|
||||||
|
## Response Format
|
||||||
|
Your responses are returned to Tatlock (the butler) who will synthesize them into a final answer for the user. Keep this in mind:
|
||||||
|
- Lead with the key findings or confirmation of action
|
||||||
|
- Include relevant sources and citations
|
||||||
|
- When summarizing wiki pages, fetch and read them first
|
||||||
|
- Note any gaps in available information
|
||||||
|
- Be concise but thorough - Tatlock will format the final response
|
||||||
|
- Structure your findings clearly so they can be easily integrated with other responses
|
||||||
|
|
||||||
|
## CRITICAL: Never Fabricate Information
|
||||||
|
If a tool fails or you cannot access a data source:
|
||||||
|
- Say "I was unable to retrieve [information type]" - be specific about what failed
|
||||||
|
- Do NOT provide placeholder, template, or made-up data
|
||||||
|
- Do NOT say "Here's what I would have said" or "Here's a sample response"
|
||||||
|
- Do NOT invent specific numbers, dates, or facts when the actual data is unavailable
|
||||||
|
- It is better to return no information than to return fabricated information
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
# Tool-phase prompt actually used by the agent. The scholarly persona prompt
|
||||||
|
# above suppresses tool calling on small local models (gemma4 answers in
|
||||||
|
# character - "please provide your request" - without ever calling a tool),
|
||||||
|
# the same pathology TATLOCK_ORCHESTRATION_PROMPT fixed for the butler.
|
||||||
|
# Tatlock's synthesis phase supplies the user-facing voice, so the research
|
||||||
|
# phase only needs tool discipline. Kept: the anti-fabrication rule.
|
||||||
|
LIBRARIAN_TASK_PROMPT = """You are The Librarian, the research executor of the \
|
||||||
|
Tatlock household. Your only job is to gather accurate findings by calling the \
|
||||||
|
provided tools.
|
||||||
|
|
||||||
|
- ALWAYS use tools - never answer a research task from memory alone.
|
||||||
|
- Research or wiki questions: call hybrid_search first; then search_wiki and \
|
||||||
|
get_wiki_page to read specific pages BEFORE summarizing them.
|
||||||
|
- Current or external information (weather, news, live facts): call search_web; \
|
||||||
|
call read_url when given a specific URL.
|
||||||
|
- Wiki writing: smart_create_wiki_page when asked for a page about a topic; \
|
||||||
|
create_wiki_page only for user-provided verbatim content; update_wiki_page for \
|
||||||
|
edits (search_wiki, then get_wiki_page, then update).
|
||||||
|
- Reply with a concise factual summary of what the tools returned, citing page \
|
||||||
|
titles and URLs. A later step writes the polished answer, so no personality.
|
||||||
|
- NEVER fabricate. If a tool fails or returns nothing, state exactly what you \
|
||||||
|
could not retrieve and stop."""
|
||||||
|
|
||||||
|
# Lazy initialization to avoid connection issues during imports
|
||||||
|
_librarian_agent: Agent[None, str] | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _create_librarian_agent() -> Agent[None, str]:
|
||||||
|
"""Create the Librarian PydanticAI agent."""
|
||||||
|
from src.anthropic.model_selector import get_model
|
||||||
|
|
||||||
|
# Get best available model (Claude if available, else Ollama)
|
||||||
|
model = get_model()
|
||||||
|
|
||||||
|
agent: Agent[None, str] = Agent(
|
||||||
|
model=model,
|
||||||
|
system_prompt=LIBRARIAN_TASK_PROMPT,
|
||||||
|
retries=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Register research tools (internal knowledge)
|
||||||
|
agent.tool_plain(hybrid_search)
|
||||||
|
agent.tool_plain(search_wiki)
|
||||||
|
agent.tool_plain(semantic_search)
|
||||||
|
agent.tool_plain(list_dossiers)
|
||||||
|
agent.tool_plain(get_dossier_pages)
|
||||||
|
agent.tool_plain(explore_knowledge_graph)
|
||||||
|
agent.tool_plain(find_related_entities)
|
||||||
|
|
||||||
|
# Register web search & content extraction tools
|
||||||
|
agent.tool_plain(search_web)
|
||||||
|
agent.tool_plain(read_url)
|
||||||
|
agent.tool_plain(read_urls_batch)
|
||||||
|
|
||||||
|
# Register wiki read tools
|
||||||
|
agent.tool_plain(get_wiki_page)
|
||||||
|
|
||||||
|
# Register wiki write tools
|
||||||
|
agent.tool_plain(create_wiki_page)
|
||||||
|
agent.tool_plain(update_wiki_page)
|
||||||
|
agent.tool_plain(smart_create_wiki_page)
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model_info
|
||||||
|
|
||||||
|
model_info = get_model_info()
|
||||||
|
logger.info(
|
||||||
|
"librarian_agent_created",
|
||||||
|
backend=model_info["backend"],
|
||||||
|
model=model_info["model"],
|
||||||
|
tool_count=14, # 7 research + 3 web + 1 wiki read + 3 wiki write
|
||||||
|
)
|
||||||
|
|
||||||
|
return agent
|
||||||
|
|
||||||
|
|
||||||
|
def get_librarian_agent() -> Agent[None, str]:
|
||||||
|
"""
|
||||||
|
Get the Librarian agent instance (lazy initialization).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PydanticAI Agent configured for research tasks
|
||||||
|
"""
|
||||||
|
global _librarian_agent
|
||||||
|
if _librarian_agent is None:
|
||||||
|
_librarian_agent = _create_librarian_agent()
|
||||||
|
return _librarian_agent
|
||||||
|
|
||||||
|
|
||||||
|
async def run_librarian(
|
||||||
|
task: str,
|
||||||
|
context: str = "",
|
||||||
|
message_history: list[Any] | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Execute a research task with The Librarian.
|
||||||
|
|
||||||
|
This is the main entry point for delegating research tasks
|
||||||
|
to The Librarian from Tatlock or other agents.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: The research task or question
|
||||||
|
context: Additional context from conversation
|
||||||
|
message_history: Optional conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Research results and findings
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
AgentError: If the research task fails. Exception detail is
|
||||||
|
logged here; callers map the failure to a user-safe message.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
result = await run_librarian(
|
||||||
|
task="Find information about Docker networking",
|
||||||
|
context="User is setting up a homelab",
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
agent = get_librarian_agent()
|
||||||
|
|
||||||
|
# Build prompt with context if provided
|
||||||
|
prompt = task
|
||||||
|
if context:
|
||||||
|
prompt = f"Context: {context}\n\nTask: {task}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"librarian_task_started",
|
||||||
|
task=task[:100],
|
||||||
|
has_context=bool(context),
|
||||||
|
has_history=bool(message_history),
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# One shared library-desk connection for all tool calls in this run
|
||||||
|
async with library_client_session():
|
||||||
|
result = await agent.run(
|
||||||
|
prompt,
|
||||||
|
message_history=message_history,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"librarian_task_completed",
|
||||||
|
task=task[:50],
|
||||||
|
output_length=len(result.output),
|
||||||
|
)
|
||||||
|
|
||||||
|
return result.output
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
# Full detail stays in the logs; callers receive a structured
|
||||||
|
# failure instead of error text masquerading as research output.
|
||||||
|
logger.error(
|
||||||
|
"librarian_task_error",
|
||||||
|
task=task[:50],
|
||||||
|
error=str(e),
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
raise AgentError("Research task failed", agent_name="librarian") from e
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
"""
|
||||||
|
Librarian capability registration for the Household Registry.
|
||||||
|
|
||||||
|
Defines The Librarian's capabilities and registers it as a
|
||||||
|
household member for coordination by the Steward and Tatlock.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.agents.librarian.agent import get_librarian_agent
|
||||||
|
from src.agents.librarian.tools import LIBRARIAN_TOOLS
|
||||||
|
from src.core.household_registry import (
|
||||||
|
HouseholdCapability,
|
||||||
|
get_household_registry,
|
||||||
|
)
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# The Librarian's capability summary for Steward coordination
|
||||||
|
LIBRARIAN_CAPABILITY = HouseholdCapability(
|
||||||
|
name="librarian",
|
||||||
|
role="The Librarian",
|
||||||
|
category="research",
|
||||||
|
description=(
|
||||||
|
"Research, web search, and wiki management: can SEARCH the web for current "
|
||||||
|
"information, READ URLs/articles, CREATE wiki pages about topics "
|
||||||
|
"(with automatic HybridRAG research), UPDATE existing pages, "
|
||||||
|
"and synthesize information from multiple sources. "
|
||||||
|
"Use for: 'search for X', 'what is X', 'create a page about X', 'read this URL'"
|
||||||
|
),
|
||||||
|
domains=[
|
||||||
|
"research",
|
||||||
|
"knowledge",
|
||||||
|
"information",
|
||||||
|
"wiki",
|
||||||
|
"documents",
|
||||||
|
"search",
|
||||||
|
"web",
|
||||||
|
"url",
|
||||||
|
"internet",
|
||||||
|
"synthesis",
|
||||||
|
"create",
|
||||||
|
"write",
|
||||||
|
"update",
|
||||||
|
],
|
||||||
|
cost="medium", # Multiple API calls to library-desk
|
||||||
|
requires_network=True, # Needs library-desk API access
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_librarian_capability() -> HouseholdCapability:
|
||||||
|
"""Get The Librarian's capability definition."""
|
||||||
|
return LIBRARIAN_CAPABILITY
|
||||||
|
|
||||||
|
|
||||||
|
def register_librarian() -> None:
|
||||||
|
"""
|
||||||
|
Register The Librarian with the Household Registry.
|
||||||
|
|
||||||
|
This makes The Librarian available for:
|
||||||
|
- Steward recommendations (via capability summary)
|
||||||
|
- Tatlock delegation (via agent reference)
|
||||||
|
- Tool scoping (via tool list)
|
||||||
|
"""
|
||||||
|
registry = get_household_registry()
|
||||||
|
|
||||||
|
# Check if already registered
|
||||||
|
if "librarian" in registry:
|
||||||
|
logger.debug("librarian_already_registered")
|
||||||
|
return
|
||||||
|
|
||||||
|
registry.register(
|
||||||
|
name="librarian",
|
||||||
|
capability=LIBRARIAN_CAPABILITY,
|
||||||
|
tools=LIBRARIAN_TOOLS,
|
||||||
|
agent=get_librarian_agent(),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"librarian_registered",
|
||||||
|
role=LIBRARIAN_CAPABILITY.role,
|
||||||
|
domains=LIBRARIAN_CAPABILITY.domains,
|
||||||
|
tool_count=len(LIBRARIAN_TOOLS),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def unregister_librarian() -> None:
|
||||||
|
"""Unregister The Librarian from the Household Registry."""
|
||||||
|
registry = get_household_registry()
|
||||||
|
registry.unregister("librarian")
|
||||||
|
logger.info("librarian_unregistered")
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+25
-48
@@ -12,16 +12,16 @@ infrastructure is real production code.
|
|||||||
import asyncio
|
import asyncio
|
||||||
import random
|
import random
|
||||||
import secrets
|
import secrets
|
||||||
from typing import AsyncGenerator, Any
|
from collections.abc import AsyncGenerator
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from src.agents.base import AgentInterface, OutputItem
|
from src.agents.base import AgentInterface, OutputItem
|
||||||
from src.core.exceptions import (
|
from src.core.exceptions import (
|
||||||
RateLimitError,
|
|
||||||
ContextLengthError,
|
|
||||||
APIError,
|
APIError,
|
||||||
|
ContextLengthError,
|
||||||
|
RateLimitError,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# Mock lorem ipsum content
|
# Mock lorem ipsum content
|
||||||
LOREM_PARAGRAPHS = [
|
LOREM_PARAGRAPHS = [
|
||||||
"Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed do eiusmod tempor incididunt ut labore et dolore magna aliqua.",
|
"Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed do eiusmod tempor incididunt ut labore et dolore magna aliqua.",
|
||||||
@@ -46,33 +46,27 @@ MOCK_TOOLS = [
|
|||||||
"description": "Search the knowledge base for relevant information",
|
"description": "Search the knowledge base for relevant information",
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {"query": {"type": "string", "description": "Search query"}},
|
||||||
"query": {"type": "string", "description": "Search query"}
|
"required": ["query"],
|
||||||
},
|
},
|
||||||
"required": ["query"]
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "calculate",
|
"name": "calculate",
|
||||||
"description": "Perform mathematical calculations",
|
"description": "Perform mathematical calculations",
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {"expression": {"type": "string", "description": "Math expression"}},
|
||||||
"expression": {"type": "string", "description": "Math expression"}
|
"required": ["expression"],
|
||||||
},
|
},
|
||||||
"required": ["expression"]
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "get_weather",
|
"name": "get_weather",
|
||||||
"description": "Get current weather for a location",
|
"description": "Get current weather for a location",
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {"location": {"type": "string", "description": "City name"}},
|
||||||
"location": {"type": "string", "description": "City name"}
|
"required": ["location"],
|
||||||
},
|
},
|
||||||
"required": ["location"]
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -109,7 +103,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
temperature: float = 1.0,
|
temperature: float = 1.0,
|
||||||
max_tokens: int | None = None,
|
max_tokens: int | None = None,
|
||||||
stop: list[str] | None = None,
|
stop: list[str] | None = None,
|
||||||
**kwargs: Any
|
**kwargs: Any,
|
||||||
) -> AsyncGenerator[OutputItem, None]:
|
) -> AsyncGenerator[OutputItem, None]:
|
||||||
"""
|
"""
|
||||||
Generate mock response with reasoning, tools, and content.
|
Generate mock response with reasoning, tools, and content.
|
||||||
@@ -123,8 +117,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
# 1. Yield reasoning item if requested
|
# 1. Yield reasoning item if requested
|
||||||
if reasoning and reasoning.get("summary") == "auto":
|
if reasoning and reasoning.get("summary") == "auto":
|
||||||
yield await self._create_reasoning_item(
|
yield await self._create_reasoning_item(
|
||||||
messages,
|
messages, effort=reasoning.get("effort", "medium")
|
||||||
effort=reasoning.get("effort", "medium")
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# 2. Randomly yield function calls if tools available (30% chance)
|
# 2. Randomly yield function calls if tools available (30% chance)
|
||||||
@@ -150,7 +143,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
"reasoning": True,
|
"reasoning": True,
|
||||||
"tools": True,
|
"tools": True,
|
||||||
"vision": False, # Not yet
|
"vision": False, # Not yet
|
||||||
"audio": False, # Not yet
|
"audio": False, # Not yet
|
||||||
}
|
}
|
||||||
|
|
||||||
# Private helper methods
|
# Private helper methods
|
||||||
@@ -179,9 +172,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
raise APIError("Invalid tool call: tool 'nonexistent' not found (mock trigger)")
|
raise APIError("Invalid tool call: tool 'nonexistent' not found (mock trigger)")
|
||||||
|
|
||||||
async def _create_reasoning_item(
|
async def _create_reasoning_item(
|
||||||
self,
|
self, messages: list[dict], effort: str = "medium"
|
||||||
messages: list[dict],
|
|
||||||
effort: str = "medium"
|
|
||||||
) -> OutputItem:
|
) -> OutputItem:
|
||||||
"""Create a reasoning output item with mock thinking steps."""
|
"""Create a reasoning output item with mock thinking steps."""
|
||||||
|
|
||||||
@@ -200,23 +191,17 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
steps = random.sample(REASONING_STEPS, min(num_steps, len(REASONING_STEPS)))
|
steps = random.sample(REASONING_STEPS, min(num_steps, len(REASONING_STEPS)))
|
||||||
|
|
||||||
return OutputItem(
|
return OutputItem(
|
||||||
type="reasoning",
|
type="reasoning", id=f"rs_{generate_id()}", summary=steps, status="completed"
|
||||||
id=f"rs_{generate_id()}",
|
|
||||||
summary=steps,
|
|
||||||
status="completed"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
async def _create_tool_calls(
|
async def _create_tool_calls(self, tools: list[dict]) -> AsyncGenerator[OutputItem, None]:
|
||||||
self,
|
|
||||||
tools: list[dict]
|
|
||||||
) -> AsyncGenerator[OutputItem, None]:
|
|
||||||
"""Create mock function call output items."""
|
"""Create mock function call output items."""
|
||||||
|
|
||||||
# Randomly select 1-2 tools to "call"
|
# Randomly select 1-2 tools to "call"
|
||||||
num_calls = random.randint(1, 2)
|
num_calls = random.randint(1, 2)
|
||||||
selected_tools = random.sample(
|
selected_tools = random.sample(
|
||||||
MOCK_TOOLS[:min(len(MOCK_TOOLS), len(tools))],
|
MOCK_TOOLS[: min(len(MOCK_TOOLS), len(tools))],
|
||||||
min(num_calls, len(MOCK_TOOLS), len(tools))
|
min(num_calls, len(MOCK_TOOLS), len(tools)),
|
||||||
)
|
)
|
||||||
|
|
||||||
for tool in selected_tools:
|
for tool in selected_tools:
|
||||||
@@ -228,7 +213,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
id=f"fc_{generate_id()}",
|
id=f"fc_{generate_id()}",
|
||||||
name=tool["name"],
|
name=tool["name"],
|
||||||
arguments=args,
|
arguments=args,
|
||||||
status="completed"
|
status="completed",
|
||||||
)
|
)
|
||||||
|
|
||||||
def _generate_mock_args(self, tool: dict) -> str:
|
def _generate_mock_args(self, tool: dict) -> str:
|
||||||
@@ -254,11 +239,7 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
# Generic mock arguments
|
# Generic mock arguments
|
||||||
return json.dumps({"input": "mock_value"})
|
return json.dumps({"input": "mock_value"})
|
||||||
|
|
||||||
async def _create_message_item(
|
async def _create_message_item(self, messages: list[dict], temperature: float) -> OutputItem:
|
||||||
self,
|
|
||||||
messages: list[dict],
|
|
||||||
temperature: float
|
|
||||||
) -> OutputItem:
|
|
||||||
"""Create final message output item with lorem ipsum content."""
|
"""Create final message output item with lorem ipsum content."""
|
||||||
|
|
||||||
# Select random lorem ipsum paragraphs
|
# Select random lorem ipsum paragraphs
|
||||||
@@ -270,10 +251,6 @@ class LoremTesterAgent(AgentInterface):
|
|||||||
type="message",
|
type="message",
|
||||||
id=f"msg_{generate_id()}",
|
id=f"msg_{generate_id()}",
|
||||||
role="assistant",
|
role="assistant",
|
||||||
content=[{
|
content=[{"type": "output_text", "text": content, "annotations": []}],
|
||||||
"type": "output_text",
|
status="completed",
|
||||||
"text": content,
|
|
||||||
"annotations": []
|
|
||||||
}],
|
|
||||||
status="completed"
|
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,526 @@
|
|||||||
|
"""
|
||||||
|
Orchestration module for multi-expert agent coordination.
|
||||||
|
|
||||||
|
Provides infrastructure for Tatlock to orchestrate expert agents
|
||||||
|
with streaming think updates to keep users informed of progress.
|
||||||
|
|
||||||
|
Key pattern: Stream user-facing interactions, use run() internally
|
||||||
|
to avoid Ollama streaming+tool call bugs.
|
||||||
|
|
||||||
|
Supports:
|
||||||
|
- Single expert delegation with think updates
|
||||||
|
- Sequential multi-expert execution (task A → task B → task C)
|
||||||
|
- Parallel multi-expert execution (tasks A, B, C concurrently)
|
||||||
|
- Result aggregation from multiple experts
|
||||||
|
- Partial failure handling
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from collections.abc import AsyncGenerator
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from enum import Enum
|
||||||
|
|
||||||
|
from src.agents.delegation import DelegationResult, DelegationTask, delegate_to_librarian
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class ExecutionMode(str, Enum):
|
||||||
|
"""Execution mode for multi-expert coordination."""
|
||||||
|
|
||||||
|
SEQUENTIAL = "sequential" # One at a time, in order
|
||||||
|
PARALLEL = "parallel" # All at once, concurrently
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class OrchestrationContext:
|
||||||
|
"""
|
||||||
|
Context for an orchestration session.
|
||||||
|
|
||||||
|
Tracks the user's request, delegation tasks, and results.
|
||||||
|
"""
|
||||||
|
|
||||||
|
user_message: str
|
||||||
|
steward_note: str
|
||||||
|
conversation_id: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def parse_delegation_from_steward_note(steward_note: str) -> DelegationTask | None:
|
||||||
|
"""
|
||||||
|
Parse a delegation task from Steward's note.
|
||||||
|
|
||||||
|
Looks for the DELEGATE: pattern in the Steward's recommendation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
steward_note: Formatted note from Steward
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DelegationTask if delegation found, None otherwise
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> note = "DELEGATE: librarian to create a wiki page about CI/CD"
|
||||||
|
>>> task = parse_delegation_from_steward_note(note)
|
||||||
|
>>> task.expert_name
|
||||||
|
'librarian'
|
||||||
|
>>> task.task
|
||||||
|
'create a wiki page about CI/CD'
|
||||||
|
"""
|
||||||
|
import re
|
||||||
|
|
||||||
|
# Look for DELEGATE: pattern
|
||||||
|
# Match: "DELEGATE: expert_name to action description"
|
||||||
|
match = re.search(
|
||||||
|
r"DELEGATE:\s*(\w+)\s+to\s+(.+?)(?:\n|REASON:|COMPLEXITY:|CONTEXT:|$)",
|
||||||
|
steward_note,
|
||||||
|
re.IGNORECASE | re.MULTILINE,
|
||||||
|
)
|
||||||
|
|
||||||
|
if match:
|
||||||
|
expert_name = match.group(1).lower()
|
||||||
|
task_description = match.group(2).strip()
|
||||||
|
|
||||||
|
# Handle "none" case
|
||||||
|
if expert_name == "none":
|
||||||
|
return None
|
||||||
|
|
||||||
|
return DelegationTask(
|
||||||
|
expert_name=expert_name,
|
||||||
|
task=task_description,
|
||||||
|
)
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def execute_delegation(
|
||||||
|
task: DelegationTask,
|
||||||
|
) -> DelegationResult:
|
||||||
|
"""
|
||||||
|
Execute a delegation task.
|
||||||
|
|
||||||
|
Routes to the appropriate expert agent based on expert_name.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
task: Delegation task to execute
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
DelegationResult from the expert agent
|
||||||
|
"""
|
||||||
|
logger.info(
|
||||||
|
"executing_delegation",
|
||||||
|
expert=task.expert_name,
|
||||||
|
task=task.task[:50],
|
||||||
|
)
|
||||||
|
|
||||||
|
if task.expert_name == "librarian":
|
||||||
|
return await delegate_to_librarian(
|
||||||
|
task=task.task,
|
||||||
|
context=task.context,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Future experts would be added here:
|
||||||
|
# elif task.expert_name == "memory":
|
||||||
|
# return await delegate_to_memory(task.task, task.context)
|
||||||
|
# elif task.expert_name == "home_automation":
|
||||||
|
# return await delegate_to_home_automation(task.task, task.context)
|
||||||
|
|
||||||
|
# Unknown expert - return error result
|
||||||
|
logger.warning("unknown_expert", expert=task.expert_name)
|
||||||
|
return DelegationResult(
|
||||||
|
expert_name=task.expert_name,
|
||||||
|
task=task.task,
|
||||||
|
success=False,
|
||||||
|
output="",
|
||||||
|
error=f"Unknown expert: {task.expert_name}",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def orchestrate_with_think_updates(
|
||||||
|
user_message: str,
|
||||||
|
steward_note: str,
|
||||||
|
delegation_task: DelegationTask | None = None,
|
||||||
|
) -> AsyncGenerator[str, None]:
|
||||||
|
"""
|
||||||
|
Orchestrate expert delegation with streaming think updates.
|
||||||
|
|
||||||
|
Emits <think> updates before and after delegation calls to
|
||||||
|
keep the user informed of progress. Expert calls use run()
|
||||||
|
internally to avoid Ollama streaming bugs.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_message: Original user message
|
||||||
|
steward_note: Steward's analysis and instructions
|
||||||
|
delegation_task: Optional pre-parsed delegation task
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
Think update strings and final expert output
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> async for update in orchestrate_with_think_updates(
|
||||||
|
... "Create a wiki page about CI/CD",
|
||||||
|
... "DELEGATE: librarian to create wiki page",
|
||||||
|
... ):
|
||||||
|
... print(update)
|
||||||
|
<think>Consulting The Librarian...</think>
|
||||||
|
<think>Delegation complete.</think>
|
||||||
|
[Wiki page created successfully...]
|
||||||
|
"""
|
||||||
|
# Parse delegation if not provided
|
||||||
|
if delegation_task is None:
|
||||||
|
delegation_task = parse_delegation_from_steward_note(steward_note)
|
||||||
|
|
||||||
|
if delegation_task is None:
|
||||||
|
# No delegation needed - nothing to orchestrate
|
||||||
|
logger.debug("no_delegation_needed")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Stream: About to delegate
|
||||||
|
expert_display_name = delegation_task.expert_name.title()
|
||||||
|
if delegation_task.expert_name == "librarian":
|
||||||
|
expert_display_name = "The Librarian"
|
||||||
|
|
||||||
|
yield f"🤝 Consulting {expert_display_name}...\n"
|
||||||
|
|
||||||
|
# Execute delegation (uses run() internally)
|
||||||
|
result = await execute_delegation(delegation_task)
|
||||||
|
|
||||||
|
if result.success:
|
||||||
|
yield f"✅ {expert_display_name} completed research.\n"
|
||||||
|
|
||||||
|
# Yield the expert's findings
|
||||||
|
if result.output:
|
||||||
|
yield f"\n{result.output}"
|
||||||
|
else:
|
||||||
|
yield f"⚠️ {expert_display_name} encountered an issue: {result.error}\n"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"orchestration_complete",
|
||||||
|
expert=delegation_task.expert_name,
|
||||||
|
success=result.success,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_delegation_context(
|
||||||
|
steward_note: str,
|
||||||
|
) -> dict[str, str]:
|
||||||
|
"""
|
||||||
|
Extract context fields from Steward's note.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
steward_note: Formatted note from Steward
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with reason, complexity, and context
|
||||||
|
"""
|
||||||
|
import re
|
||||||
|
|
||||||
|
result = {
|
||||||
|
"reason": "",
|
||||||
|
"complexity": "",
|
||||||
|
"context": "",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extract REASON:
|
||||||
|
reason_match = re.search(
|
||||||
|
r"REASON:\s*(.+?)(?:\n|COMPLEXITY:|CONTEXT:|$)", steward_note, re.IGNORECASE
|
||||||
|
)
|
||||||
|
if reason_match:
|
||||||
|
result["reason"] = reason_match.group(1).strip()
|
||||||
|
|
||||||
|
# Extract COMPLEXITY:
|
||||||
|
complexity_match = re.search(
|
||||||
|
r"COMPLEXITY:\s*(.+?)(?:\n|CONTEXT:|$)", steward_note, re.IGNORECASE
|
||||||
|
)
|
||||||
|
if complexity_match:
|
||||||
|
result["complexity"] = complexity_match.group(1).strip()
|
||||||
|
|
||||||
|
# Extract CONTEXT:
|
||||||
|
context_match = re.search(r"CONTEXT:\s*(.+?)$", steward_note, re.IGNORECASE | re.MULTILINE)
|
||||||
|
if context_match:
|
||||||
|
result["context"] = context_match.group(1).strip()
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# Multi-Expert Coordination
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class MultiExpertResult:
|
||||||
|
"""
|
||||||
|
Aggregated result from multiple expert delegations.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
results: Dict mapping expert name to their result
|
||||||
|
all_succeeded: True if all delegations succeeded
|
||||||
|
failed_experts: List of expert names that failed
|
||||||
|
combined_output: Aggregated output from all successful experts
|
||||||
|
"""
|
||||||
|
|
||||||
|
results: dict[str, DelegationResult] = field(default_factory=dict)
|
||||||
|
all_succeeded: bool = True
|
||||||
|
failed_experts: list[str] = field(default_factory=list)
|
||||||
|
combined_output: str = ""
|
||||||
|
|
||||||
|
def add_result(self, result: DelegationResult) -> None:
|
||||||
|
"""Add a result and update aggregation state."""
|
||||||
|
self.results[result.expert_name] = result
|
||||||
|
if not result.success:
|
||||||
|
self.all_succeeded = False
|
||||||
|
self.failed_experts.append(result.expert_name)
|
||||||
|
|
||||||
|
def aggregate_outputs(self, separator: str = "\n\n---\n\n") -> str:
|
||||||
|
"""Combine all successful outputs into one string."""
|
||||||
|
outputs = []
|
||||||
|
for expert_name, result in self.results.items():
|
||||||
|
if result.success and result.output:
|
||||||
|
outputs.append(f"**{expert_name.title()}**: {result.output}")
|
||||||
|
|
||||||
|
self.combined_output = separator.join(outputs)
|
||||||
|
return self.combined_output
|
||||||
|
|
||||||
|
|
||||||
|
async def execute_sequential(
|
||||||
|
tasks: list[DelegationTask],
|
||||||
|
stop_on_failure: bool = False,
|
||||||
|
) -> MultiExpertResult:
|
||||||
|
"""
|
||||||
|
Execute multiple delegation tasks sequentially.
|
||||||
|
|
||||||
|
Tasks run one after another in order. Later tasks can depend on
|
||||||
|
earlier results (though this function doesn't handle passing
|
||||||
|
results between tasks - that's the orchestrator's job).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tasks: List of delegation tasks to execute in order
|
||||||
|
stop_on_failure: If True, stop execution if any task fails
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
MultiExpertResult with all task results
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> tasks = [
|
||||||
|
... DelegationTask(expert_name="memory", task="get user location"),
|
||||||
|
... DelegationTask(expert_name="librarian", task="search weather"),
|
||||||
|
... ]
|
||||||
|
>>> result = await execute_sequential(tasks)
|
||||||
|
>>> result.all_succeeded
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
multi_result = MultiExpertResult()
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"sequential_execution_started",
|
||||||
|
task_count=len(tasks),
|
||||||
|
experts=[t.expert_name for t in tasks],
|
||||||
|
)
|
||||||
|
|
||||||
|
for i, task in enumerate(tasks):
|
||||||
|
logger.debug(
|
||||||
|
"sequential_task_executing",
|
||||||
|
index=i,
|
||||||
|
expert=task.expert_name,
|
||||||
|
task=task.task[:50],
|
||||||
|
)
|
||||||
|
|
||||||
|
result = await execute_delegation(task)
|
||||||
|
multi_result.add_result(result)
|
||||||
|
|
||||||
|
if not result.success and stop_on_failure:
|
||||||
|
logger.warning(
|
||||||
|
"sequential_execution_stopped",
|
||||||
|
failed_at=i,
|
||||||
|
expert=task.expert_name,
|
||||||
|
error=result.error,
|
||||||
|
)
|
||||||
|
break
|
||||||
|
|
||||||
|
multi_result.aggregate_outputs()
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"sequential_execution_complete",
|
||||||
|
total_tasks=len(tasks),
|
||||||
|
succeeded=len(tasks) - len(multi_result.failed_experts),
|
||||||
|
failed=len(multi_result.failed_experts),
|
||||||
|
)
|
||||||
|
|
||||||
|
return multi_result
|
||||||
|
|
||||||
|
|
||||||
|
async def execute_parallel(
|
||||||
|
tasks: list[DelegationTask],
|
||||||
|
) -> MultiExpertResult:
|
||||||
|
"""
|
||||||
|
Execute multiple delegation tasks in parallel.
|
||||||
|
|
||||||
|
All tasks run concurrently using asyncio.gather. Use this when
|
||||||
|
tasks are independent and don't depend on each other's results.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tasks: List of delegation tasks to execute concurrently
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
MultiExpertResult with all task results
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> tasks = [
|
||||||
|
... DelegationTask(expert_name="librarian", task="search wiki"),
|
||||||
|
... DelegationTask(expert_name="memory", task="get preferences"),
|
||||||
|
... ]
|
||||||
|
>>> result = await execute_parallel(tasks)
|
||||||
|
>>> len(result.results)
|
||||||
|
2
|
||||||
|
"""
|
||||||
|
multi_result = MultiExpertResult()
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"parallel_execution_started",
|
||||||
|
task_count=len(tasks),
|
||||||
|
experts=[t.expert_name for t in tasks],
|
||||||
|
)
|
||||||
|
|
||||||
|
# Execute all tasks concurrently
|
||||||
|
results = await asyncio.gather(
|
||||||
|
*[execute_delegation(task) for task in tasks],
|
||||||
|
return_exceptions=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Process results
|
||||||
|
for i, result in enumerate(results):
|
||||||
|
if isinstance(result, Exception):
|
||||||
|
# Handle exceptions as failed delegations
|
||||||
|
error_result = DelegationResult(
|
||||||
|
expert_name=tasks[i].expert_name,
|
||||||
|
task=tasks[i].task,
|
||||||
|
success=False,
|
||||||
|
output="",
|
||||||
|
error=str(result),
|
||||||
|
)
|
||||||
|
multi_result.add_result(error_result)
|
||||||
|
logger.error(
|
||||||
|
"parallel_task_exception",
|
||||||
|
expert=tasks[i].expert_name,
|
||||||
|
error=str(result),
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
multi_result.add_result(result)
|
||||||
|
|
||||||
|
multi_result.aggregate_outputs()
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"parallel_execution_complete",
|
||||||
|
total_tasks=len(tasks),
|
||||||
|
succeeded=len(tasks) - len(multi_result.failed_experts),
|
||||||
|
failed=len(multi_result.failed_experts),
|
||||||
|
)
|
||||||
|
|
||||||
|
return multi_result
|
||||||
|
|
||||||
|
|
||||||
|
async def orchestrate_multi_expert(
|
||||||
|
tasks: list[DelegationTask],
|
||||||
|
mode: ExecutionMode = ExecutionMode.SEQUENTIAL,
|
||||||
|
stop_on_failure: bool = False,
|
||||||
|
) -> AsyncGenerator[str, None]:
|
||||||
|
"""
|
||||||
|
Orchestrate multiple expert delegations with streaming think updates.
|
||||||
|
|
||||||
|
Emits <think> updates for each delegation phase and yields
|
||||||
|
combined results at the end.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tasks: List of delegation tasks
|
||||||
|
mode: SEQUENTIAL or PARALLEL execution
|
||||||
|
stop_on_failure: For sequential mode, stop if a task fails
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
Think updates and combined expert output
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> tasks = [
|
||||||
|
... DelegationTask(expert_name="memory", task="get location"),
|
||||||
|
... DelegationTask(expert_name="librarian", task="search weather"),
|
||||||
|
... ]
|
||||||
|
>>> async for update in orchestrate_multi_expert(tasks):
|
||||||
|
... print(update)
|
||||||
|
<think>Starting multi-expert coordination (2 tasks)...</think>
|
||||||
|
<think>Consulting Memory...</think>
|
||||||
|
<think>Memory completed.</think>
|
||||||
|
<think>Consulting The Librarian...</think>
|
||||||
|
<think>The Librarian completed.</think>
|
||||||
|
<think>All experts completed successfully.</think>
|
||||||
|
[Combined output from all experts...]
|
||||||
|
"""
|
||||||
|
if not tasks:
|
||||||
|
logger.debug("no_tasks_to_orchestrate")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Stream: Starting multi-expert coordination
|
||||||
|
yield f"🎯 Starting multi-expert coordination ({len(tasks)} tasks, {mode.value})...\n"
|
||||||
|
|
||||||
|
if mode == ExecutionMode.PARALLEL:
|
||||||
|
# Parallel execution - emit one update then run all at once
|
||||||
|
expert_names = ", ".join(_get_display_name(t.expert_name) for t in tasks)
|
||||||
|
yield f"🔄 Consulting in parallel: {expert_names}...\n"
|
||||||
|
|
||||||
|
result = await execute_parallel(tasks)
|
||||||
|
|
||||||
|
# Emit completion updates for each
|
||||||
|
for expert_name, expert_result in result.results.items():
|
||||||
|
display_name = _get_display_name(expert_name)
|
||||||
|
if expert_result.success:
|
||||||
|
yield f"✅ {display_name} completed.\n"
|
||||||
|
else:
|
||||||
|
yield f"⚠️ {display_name} failed: {expert_result.error}\n"
|
||||||
|
|
||||||
|
else:
|
||||||
|
# Sequential execution - emit updates for each task
|
||||||
|
result = MultiExpertResult()
|
||||||
|
|
||||||
|
for task in tasks:
|
||||||
|
display_name = _get_display_name(task.expert_name)
|
||||||
|
yield f"🤝 Consulting {display_name}...\n"
|
||||||
|
|
||||||
|
task_result = await execute_delegation(task)
|
||||||
|
result.add_result(task_result)
|
||||||
|
|
||||||
|
if task_result.success:
|
||||||
|
yield f"✅ {display_name} completed.\n"
|
||||||
|
else:
|
||||||
|
yield f"⚠️ {display_name} failed: {task_result.error}\n"
|
||||||
|
if stop_on_failure:
|
||||||
|
yield "🛑 Stopping due to failure.\n"
|
||||||
|
break
|
||||||
|
|
||||||
|
result.aggregate_outputs()
|
||||||
|
|
||||||
|
# Stream: Summary
|
||||||
|
if result.all_succeeded:
|
||||||
|
yield "🎉 All experts completed successfully.\n"
|
||||||
|
else:
|
||||||
|
failed_names = ", ".join(_get_display_name(e) for e in result.failed_experts)
|
||||||
|
yield f"⚠️ Some experts failed: {failed_names}\n"
|
||||||
|
|
||||||
|
# Yield combined output
|
||||||
|
if result.combined_output:
|
||||||
|
yield f"\n{result.combined_output}"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"multi_expert_orchestration_complete",
|
||||||
|
task_count=len(tasks),
|
||||||
|
mode=mode.value,
|
||||||
|
all_succeeded=result.all_succeeded,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_display_name(expert_name: str) -> str:
|
||||||
|
"""Get user-friendly display name for an expert."""
|
||||||
|
display_names = {
|
||||||
|
"librarian": "The Librarian",
|
||||||
|
"memory": "Memory",
|
||||||
|
"home_automation": "Home Automation",
|
||||||
|
"tatlock_core": "Core Tools",
|
||||||
|
}
|
||||||
|
return display_names.get(expert_name, expert_name.title())
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
"""
|
||||||
|
Agent error protocol.
|
||||||
|
|
||||||
|
Structured exceptions raised by expert agents (e.g. The Librarian) so
|
||||||
|
callers - the delegation wrappers in src/agents/delegation.py - can
|
||||||
|
report success=False and map failures to curated user-safe messages
|
||||||
|
while exception detail stays in the logs.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
class AgentError(Exception):
|
||||||
|
"""Base exception for agent errors."""
|
||||||
|
|
||||||
|
def __init__(self, message: str, agent_name: str = "unknown"):
|
||||||
|
self.message = message
|
||||||
|
self.agent_name = agent_name
|
||||||
|
super().__init__(f"[{agent_name}] {message}")
|
||||||
+11
-12
@@ -8,9 +8,6 @@ It provides a central place to:
|
|||||||
- Check model capabilities
|
- Check model capabilities
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import time
|
|
||||||
from typing import Type
|
|
||||||
|
|
||||||
from src.agents.base import AgentInterface
|
from src.agents.base import AgentInterface
|
||||||
from src.agents.lorem_tester import LoremTesterAgent
|
from src.agents.lorem_tester import LoremTesterAgent
|
||||||
from src.agents.tatlock import TatlockAgent
|
from src.agents.tatlock import TatlockAgent
|
||||||
@@ -63,7 +60,7 @@ class ModelRegistry:
|
|||||||
if model_id not in cls.MODELS:
|
if model_id not in cls.MODELS:
|
||||||
raise ModelNotFoundError(model_id)
|
raise ModelNotFoundError(model_id)
|
||||||
|
|
||||||
agent_class: Type[AgentInterface] = cls.MODELS[model_id]["agent_class"]
|
agent_class: type[AgentInterface] = cls.MODELS[model_id]["agent_class"]
|
||||||
return agent_class()
|
return agent_class()
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -106,14 +103,16 @@ class ModelRegistry:
|
|||||||
agent = cls.get_agent(model_id)
|
agent = cls.get_agent(model_id)
|
||||||
capabilities = await agent.get_capabilities()
|
capabilities = await agent.get_capabilities()
|
||||||
|
|
||||||
models.append({
|
models.append(
|
||||||
"id": model_id,
|
{
|
||||||
"object": "model",
|
"id": model_id,
|
||||||
"created": config["created"],
|
"object": "model",
|
||||||
"owned_by": config["owned_by"],
|
"created": config["created"],
|
||||||
"capabilities": capabilities,
|
"owned_by": config["owned_by"],
|
||||||
"description": config["description"],
|
"capabilities": capabilities,
|
||||||
})
|
"description": config["description"],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
return models
|
return models
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ Steward agent package.
|
|||||||
The Steward analyzes incoming requests and recommends relevant household
|
The Steward analyzes incoming requests and recommends relevant household
|
||||||
capabilities, creating a two-tier architecture with the Butler.
|
capabilities, creating a two-tier architecture with the Butler.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from .agent import StewardAgent, get_steward_agent
|
from .agent import StewardAgent, get_steward_agent
|
||||||
from .schemas import ConversationContext, StewardRecommendation
|
from .schemas import ConversationContext, StewardRecommendation
|
||||||
from .service import analyze_request, format_steward_note
|
from .service import analyze_request, format_steward_note
|
||||||
|
|||||||
+152
-48
@@ -5,11 +5,13 @@ The Steward analyzes incoming requests, identifies relevant household
|
|||||||
capabilities, and provides focused recommendations to Tatlock (the Butler).
|
capabilities, and provides focused recommendations to Tatlock (the Butler).
|
||||||
This creates a two-tier architecture that prevents cognitive overload.
|
This creates a two-tier architecture that prevents cognitive overload.
|
||||||
|
|
||||||
Uses plain text output (not JSON) for reliability with Ollama models.
|
Uses plain text output (not JSON) for reliability. Supports both Claude
|
||||||
|
(preferred) and Ollama (fallback) backends via direct API calls.
|
||||||
"""
|
"""
|
||||||
import httpx
|
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model_info, is_claude_available, resolve_backend
|
||||||
from src.core.config import config
|
from src.core.config import config
|
||||||
from src.core.household_registry import get_household_registry
|
from src.core.household_registry import get_household_registry
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
@@ -27,9 +29,7 @@ def build_steward_prompt(query: str, conversation_history: list[dict]) -> str:
|
|||||||
|
|
||||||
cap_list = []
|
cap_list = []
|
||||||
for cap in capabilities:
|
for cap in capabilities:
|
||||||
cap_list.append(
|
cap_list.append(f"• {cap.name} - {cap.description} (domains: {', '.join(cap.domains)})")
|
||||||
f"• {cap.name} - {cap.description} (domains: {', '.join(cap.domains)})"
|
|
||||||
)
|
|
||||||
capabilities_text = "\n".join(cap_list)
|
capabilities_text = "\n".join(cap_list)
|
||||||
|
|
||||||
# Format conversation history if present
|
# Format conversation history if present
|
||||||
@@ -48,7 +48,7 @@ AVAILABLE HOUSEHOLD CAPABILITIES:
|
|||||||
{capabilities_text}
|
{capabilities_text}
|
||||||
|
|
||||||
YOUR TASK:
|
YOUR TASK:
|
||||||
Analyze the user's query and recommend which capabilities are needed.
|
Analyze the user's query and recommend which capabilities are needed, with specific delegation instructions.
|
||||||
{history_text}
|
{history_text}
|
||||||
|
|
||||||
USER QUERY: {query}
|
USER QUERY: {query}
|
||||||
@@ -56,19 +56,45 @@ USER QUERY: {query}
|
|||||||
GUIDELINES:
|
GUIDELINES:
|
||||||
- Be conservative - only recommend truly necessary capabilities
|
- Be conservative - only recommend truly necessary capabilities
|
||||||
- Simple greetings/chat → no capabilities needed (conversational response only)
|
- Simple greetings/chat → no capabilities needed (conversational response only)
|
||||||
|
- Questions about prior conversation ("what did I say", "what we discussed") → no capabilities (Tatlock has full history)
|
||||||
- Math/calculations → tatlock_core
|
- Math/calculations → tatlock_core
|
||||||
- Web searches → tatlock_core
|
|
||||||
- Time/date queries → tatlock_core
|
- Time/date queries → tatlock_core
|
||||||
|
- PERSONAL MEMORY queries → biographer to recall (ALWAYS use for questions about the user themselves):
|
||||||
|
- "where do I live", "what's my location", "my address" → biographer to recall location
|
||||||
|
- "what's my name", "who am I" → biographer to recall name
|
||||||
|
- "what car do I drive", "my vehicle" → biographer to recall car
|
||||||
|
- "what do you know about me", "what have I told you" → biographer to recall or list_memories
|
||||||
|
- "remember that I...", "store that..." → biographer to store_insight
|
||||||
|
- "forget my...", "delete..." → biographer to forget_memory
|
||||||
|
- "my timezone", "my preferences" → biographer to recall preferences
|
||||||
|
- Web searches, weather, news, current information → librarian with search_web
|
||||||
|
- Read a URL or article → librarian with read_url
|
||||||
|
- Wiki creation ("create a page about X", "add X to wiki") → librarian with smart_create
|
||||||
|
- Wiki updates ("update the page", "add to dossier") → librarian with update
|
||||||
|
- Research queries about TOPICS (not about the user) → librarian with hybrid_search
|
||||||
|
- In-depth research, knowledge synthesis, document lookup → librarian with hybrid_search
|
||||||
- If conversation history is relevant, note which previous turns matter
|
- If conversation history is relevant, note which previous turns matter
|
||||||
- Assess complexity: simple (1 tool), moderate (2-3 tools), complex (multiple steps)
|
- Assess complexity: simple (1 tool), moderate (2-3 tools), complex (multiple steps)
|
||||||
- If capabilities are missing, mention what would be needed
|
|
||||||
|
|
||||||
RESPOND WITH 2-3 SENTENCES:
|
RESPOND IN THIS FORMAT:
|
||||||
1. Which capabilities (if any) are needed and why
|
DELEGATE: [capability name] to [action] [specific task]
|
||||||
2. Complexity assessment (simple/moderate/complex)
|
REASON: [why this capability handles the request]
|
||||||
3. Any conversation context or missing capabilities
|
COMPLEXITY: [simple/moderate/complex]
|
||||||
|
CONTEXT: [any relevant conversation context, or "none"]
|
||||||
|
|
||||||
Use capability names in your response (e.g., "tatlock_core for calculations").
|
EXAMPLES:
|
||||||
|
- "DELEGATE: biographer to recall the user's location" (for "where do I live?")
|
||||||
|
- "DELEGATE: biographer to recall the user's car" (for "what car do I drive?")
|
||||||
|
- "DELEGATE: biographer to list_memories about the user" (for "what do you know about me?")
|
||||||
|
- "DELEGATE: biographer to store_insight about user's pet" (for "remember that I have a dog named Max")
|
||||||
|
- "DELEGATE: librarian to search_web for tomorrow's weather forecast"
|
||||||
|
- "DELEGATE: librarian to create a wiki page about CI/CD pipelines"
|
||||||
|
- "DELEGATE: librarian to hybrid_search for information about Docker networking"
|
||||||
|
- "DELEGATE: librarian to read_url https://example.com/article"
|
||||||
|
- "DELEGATE: tatlock_core to calculate the result"
|
||||||
|
- "DELEGATE: none (conversational response only)"
|
||||||
|
|
||||||
|
Be specific about what Tatlock should delegate - include the action verb (create, update, search, etc.).
|
||||||
Plain text only - no JSON, no special formatting."""
|
Plain text only - no JSON, no special formatting."""
|
||||||
|
|
||||||
|
|
||||||
@@ -79,30 +105,82 @@ class StewardAgent:
|
|||||||
Analyzes requests with full conversation context and recommends
|
Analyzes requests with full conversation context and recommends
|
||||||
which household capabilities the Butler should use.
|
which household capabilities the Butler should use.
|
||||||
|
|
||||||
Uses plain text output for reliability with Ollama models.
|
Uses plain text output for reliability. Supports both Claude
|
||||||
|
(preferred) and Ollama (fallback) backends via direct API calls.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self):
|
def __init__(self) -> None:
|
||||||
"""Initialize Steward with Ollama model (same as Tatlock for VRAM efficiency)."""
|
"""Initialize Steward with backend selection based on availability."""
|
||||||
self.ollama_host = str(config.OLLAMA_HOST).rstrip('/')
|
# Ollama config (primary)
|
||||||
self.model_name = config.OLLAMA_DEFAULT_MODEL
|
self.ollama_host = str(config.OLLAMA_HOST).rstrip("/")
|
||||||
self.timeout = 30.0 # 30 second timeout for analysis
|
self.ollama_model = config.OLLAMA_DEFAULT_MODEL
|
||||||
|
|
||||||
|
# Claude config (fallback)
|
||||||
|
self.claude_model = config.ANTHROPIC_MODEL
|
||||||
|
self._anthropic_client = None
|
||||||
|
|
||||||
|
# Determine which backend to use (Ollama-first, Claude when
|
||||||
|
# preferred via config or when Ollama is down)
|
||||||
|
self._use_claude = resolve_backend() == "claude"
|
||||||
|
|
||||||
|
self.timeout = float(config.STEWARD_TIMEOUT)
|
||||||
|
|
||||||
|
model_info = get_model_info()
|
||||||
logger.info(
|
logger.info(
|
||||||
"steward_agent_created",
|
"steward_agent_created",
|
||||||
ollama_host=self.ollama_host,
|
backend=model_info["backend"],
|
||||||
model=self.model_name,
|
model=model_info["model"],
|
||||||
timeout=self.timeout,
|
timeout=self.timeout,
|
||||||
)
|
)
|
||||||
|
|
||||||
async def analyze(
|
def _get_anthropic_client(self):
|
||||||
self,
|
"""Get or create Anthropic client (lazy initialization)."""
|
||||||
query: str,
|
if self._anthropic_client is None:
|
||||||
conversation_history: Optional[list[dict]] = None
|
from anthropic import AsyncAnthropic
|
||||||
) -> str:
|
|
||||||
|
self._anthropic_client = AsyncAnthropic(api_key=config.ANTHROPIC_API_KEY)
|
||||||
|
return self._anthropic_client
|
||||||
|
|
||||||
|
async def _call_claude(self, system_prompt: str, user_message: str) -> str:
|
||||||
|
"""Call Claude API directly for plain text generation."""
|
||||||
|
client = self._get_anthropic_client()
|
||||||
|
|
||||||
|
# No temperature: rejected by Claude Sonnet 5+ (sampling params deprecated)
|
||||||
|
response = await client.messages.create(
|
||||||
|
model=self.claude_model,
|
||||||
|
max_tokens=1024,
|
||||||
|
system=system_prompt,
|
||||||
|
messages=[{"role": "user", "content": user_message}],
|
||||||
|
)
|
||||||
|
|
||||||
|
return response.content[0].text.strip()
|
||||||
|
|
||||||
|
async def _call_ollama(self, prompt: str) -> str:
|
||||||
|
"""Call Ollama API directly for plain text generation."""
|
||||||
|
async with httpx.AsyncClient(timeout=self.timeout) as client:
|
||||||
|
response = await client.post(
|
||||||
|
f"{self.ollama_host}/api/generate",
|
||||||
|
json={
|
||||||
|
"model": self.ollama_model,
|
||||||
|
"prompt": prompt,
|
||||||
|
"stream": False,
|
||||||
|
"options": {
|
||||||
|
"temperature": 0.3, # Lower = more consistent
|
||||||
|
"top_p": 0.9,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
response.raise_for_status()
|
||||||
|
result = response.json()
|
||||||
|
return result["response"].strip()
|
||||||
|
|
||||||
|
async def analyze(self, query: str, conversation_history: list[dict] | None = None) -> str:
|
||||||
"""
|
"""
|
||||||
Analyze query and return plain text recommendation.
|
Analyze query and return plain text recommendation.
|
||||||
|
|
||||||
|
Uses Claude if available, falls back to Ollama.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
query: User's query to analyze
|
query: User's query to analyze
|
||||||
conversation_history: Previous conversation turns
|
conversation_history: Previous conversation turns
|
||||||
@@ -118,35 +196,61 @@ class StewardAgent:
|
|||||||
history = conversation_history or []
|
history = conversation_history or []
|
||||||
prompt = build_steward_prompt(query, history)
|
prompt = build_steward_prompt(query, history)
|
||||||
|
|
||||||
logger.debug("steward_calling_ollama", query_preview=query[:100])
|
backend = "claude" if self._use_claude else "ollama"
|
||||||
|
logger.debug(
|
||||||
|
"steward_calling_llm",
|
||||||
|
backend=backend,
|
||||||
|
query_preview=query[:100],
|
||||||
|
)
|
||||||
|
|
||||||
# Call Ollama API directly (more reliable than PydanticAI for plain text)
|
try:
|
||||||
async with httpx.AsyncClient(timeout=self.timeout) as client:
|
if self._use_claude:
|
||||||
response = await client.post(
|
# For Claude, split into system + user message
|
||||||
f"{self.ollama_host}/api/generate",
|
# The prompt contains both, but Claude prefers explicit system
|
||||||
json={
|
analysis_text = await self._call_claude(
|
||||||
"model": self.model_name,
|
system_prompt="You are the Steward of the household, advising the Butler (Tatlock) on which capabilities to use. Be concise and specific.",
|
||||||
"prompt": prompt,
|
user_message=prompt,
|
||||||
"stream": False,
|
)
|
||||||
"options": {
|
else:
|
||||||
"temperature": 0.3, # Lower = more consistent
|
analysis_text = await self._call_ollama(prompt)
|
||||||
"top_p": 0.9
|
|
||||||
}
|
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
response.raise_for_status()
|
|
||||||
result = response.json()
|
|
||||||
|
|
||||||
analysis_text = result["response"].strip()
|
|
||||||
|
|
||||||
logger.debug(
|
logger.debug(
|
||||||
"steward_analysis_received",
|
"steward_analysis_received",
|
||||||
text_preview=analysis_text[:150]
|
backend=backend,
|
||||||
|
text_preview=analysis_text[:150],
|
||||||
)
|
)
|
||||||
|
|
||||||
return analysis_text
|
return analysis_text
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
# Mid-request fallback: retry on the other backend when possible
|
||||||
|
if self._use_claude:
|
||||||
|
logger.warning(
|
||||||
|
"steward_claude_fallback",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
analysis_text = await self._call_ollama(prompt)
|
||||||
|
fallback_backend = "ollama_fallback"
|
||||||
|
elif is_claude_available():
|
||||||
|
logger.warning(
|
||||||
|
"steward_ollama_fallback",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
analysis_text = await self._call_claude(
|
||||||
|
system_prompt="You are the Steward of the household, advising the Butler (Tatlock) on which capabilities to use. Be concise and specific.",
|
||||||
|
user_message=prompt,
|
||||||
|
)
|
||||||
|
fallback_backend = "claude_fallback"
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"steward_analysis_received",
|
||||||
|
backend=fallback_backend,
|
||||||
|
text_preview=analysis_text[:150],
|
||||||
|
)
|
||||||
|
return analysis_text
|
||||||
|
|
||||||
|
|
||||||
# Global Steward instance
|
# Global Steward instance
|
||||||
_steward_agent = None
|
_steward_agent = None
|
||||||
|
|||||||
@@ -4,7 +4,8 @@ Steward agent schemas.
|
|||||||
Defines the structured output models for Steward's request analysis
|
Defines the structured output models for Steward's request analysis
|
||||||
and capability recommendations.
|
and capability recommendations.
|
||||||
"""
|
"""
|
||||||
from typing import Literal, Optional
|
|
||||||
|
from typing import Any, Literal
|
||||||
|
|
||||||
from pydantic import BaseModel, Field
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
@@ -16,16 +17,16 @@ class ConversationContext(BaseModel):
|
|||||||
The Steward analyzes the full conversation to identify references
|
The Steward analyzes the full conversation to identify references
|
||||||
to previous topics, helping the Butler maintain context.
|
to previous topics, helping the Butler maintain context.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
has_previous_context: bool = Field(
|
has_previous_context: bool = Field(
|
||||||
description="Whether the current request references previous conversation turns"
|
description="Whether the current request references previous conversation turns"
|
||||||
)
|
)
|
||||||
relevant_turns: list[int] = Field(
|
relevant_turns: list[int] = Field(
|
||||||
default_factory=list,
|
default_factory=list,
|
||||||
description="0-indexed turn numbers that are relevant to the current request"
|
description="0-indexed turn numbers that are relevant to the current request",
|
||||||
)
|
)
|
||||||
context_summary: str = Field(
|
context_summary: str = Field(
|
||||||
default="",
|
default="", description="Brief summary of relevant context for the Butler"
|
||||||
description="Brief summary of relevant context for the Butler"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -40,21 +41,28 @@ class StewardRecommendation(BaseModel):
|
|||||||
- Conversation context
|
- Conversation context
|
||||||
- Missing capabilities (if any)
|
- Missing capabilities (if any)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
recommended_capabilities: list[str] = Field(
|
recommended_capabilities: list[str] = Field(
|
||||||
description="List of household member names to include (e.g., ['tatlock_core'])"
|
description="List of household member names to include (e.g., ['tatlock_core'])"
|
||||||
)
|
)
|
||||||
reasoning: str = Field(
|
reasoning: str = Field(description="Explanation of why these capabilities were recommended")
|
||||||
description="Explanation of why these capabilities were recommended"
|
|
||||||
)
|
|
||||||
estimated_complexity: Literal["simple", "moderate", "complex"] = Field(
|
estimated_complexity: Literal["simple", "moderate", "complex"] = Field(
|
||||||
description="Complexity assessment: simple (1 tool), moderate (2-3 tools), complex (multiple tools/steps)"
|
description="Complexity assessment: simple (1 tool), moderate (2-3 tools), complex (multiple tools/steps)"
|
||||||
)
|
)
|
||||||
conversation_context: ConversationContext = Field(
|
conversation_context: ConversationContext = Field(
|
||||||
description="Contextual information from conversation history"
|
description="Contextual information from conversation history"
|
||||||
)
|
)
|
||||||
missing_capabilities: Optional[str] = Field(
|
missing_capabilities: str | None = Field(
|
||||||
default=None,
|
default=None,
|
||||||
description="Description of capabilities that would be helpful but aren't available"
|
description="Description of capabilities that would be helpful but aren't available",
|
||||||
|
)
|
||||||
|
memory_context: dict[str, Any] = Field(
|
||||||
|
default_factory=dict,
|
||||||
|
description="Pre-fetched user context from memory (profile, preferences)",
|
||||||
|
)
|
||||||
|
enriched_query: str = Field(
|
||||||
|
default="",
|
||||||
|
description="User query with auto-filled context (location, timezone) when not specified",
|
||||||
)
|
)
|
||||||
|
|
||||||
def format_for_butler(self) -> str:
|
def format_for_butler(self) -> str:
|
||||||
@@ -88,6 +96,34 @@ class StewardRecommendation(BaseModel):
|
|||||||
if self.missing_capabilities:
|
if self.missing_capabilities:
|
||||||
lines.append(f"⚠️ Missing: {self.missing_capabilities}")
|
lines.append(f"⚠️ Missing: {self.missing_capabilities}")
|
||||||
|
|
||||||
|
# Memory context (user profile and preferences)
|
||||||
|
if self.memory_context:
|
||||||
|
profile = self.memory_context.get("profile", {})
|
||||||
|
preferences = self.memory_context.get("preferences", {})
|
||||||
|
|
||||||
|
if profile or preferences:
|
||||||
|
lines.append("-" * 40)
|
||||||
|
lines.append("User Context:")
|
||||||
|
|
||||||
|
if profile:
|
||||||
|
for key, value in profile.items():
|
||||||
|
lines.append(f" • {key}: {value}")
|
||||||
|
|
||||||
|
if preferences:
|
||||||
|
prefs_str = ", ".join(f"{k}={v}" for k, v in preferences.items())
|
||||||
|
lines.append(f" • preferences: {prefs_str}")
|
||||||
|
|
||||||
|
# Add delegation instructions when expert agents are recommended
|
||||||
|
delegation_agents = [
|
||||||
|
c for c in self.recommended_capabilities if c in ("biographer", "librarian")
|
||||||
|
]
|
||||||
|
if delegation_agents:
|
||||||
|
lines.append("-" * 40)
|
||||||
|
lines.append("DELEGATION REQUIRED:")
|
||||||
|
for agent in delegation_agents:
|
||||||
|
lines.append(f' Call: delegate_to_{agent}(task="[user request]")')
|
||||||
|
lines.append(f' Or output: [DELEGATE:{agent}] task="[user request]"')
|
||||||
|
|
||||||
lines.append("=" * 40)
|
lines.append("=" * 40)
|
||||||
|
|
||||||
return "\n".join(lines)
|
return "\n".join(lines)
|
||||||
|
|||||||
+276
-65
@@ -1,52 +1,108 @@
|
|||||||
"""
|
"""
|
||||||
Steward service layer.
|
Steward service layer.
|
||||||
|
|
||||||
Provides high-level interface for request analysis with logging,
|
Provides high-level interface for request analysis with logging
|
||||||
benchmarking, and error handling.
|
and error handling.
|
||||||
|
|
||||||
Parses plain text recommendations into structured data.
|
Parses plain text recommendations into structured data.
|
||||||
|
Includes memory pre-fetch for user context injection.
|
||||||
"""
|
"""
|
||||||
import re
|
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from src.core.benchmarks import PerformanceBenchmark, get_benchmark_store
|
import re
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from src.core.household_registry import get_household_registry
|
from src.core.household_registry import get_household_registry
|
||||||
from src.core.logging_config import get_logger, log_operation
|
from src.core.logging_config import get_logger, log_operation
|
||||||
|
from src.core.memory_service import memory_service
|
||||||
|
|
||||||
from .agent import get_steward_agent
|
from .agent import get_steward_agent
|
||||||
from .schemas import ConversationContext, StewardRecommendation
|
from .schemas import ConversationContext, StewardRecommendation
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
_DELEGATE_LINE_RE = re.compile(r"^[ \t]*DELEGATE:[ \t]*(.+)$", re.IGNORECASE | re.MULTILINE)
|
||||||
|
|
||||||
|
|
||||||
|
def _mentions(needle: str, haystack: str) -> bool:
|
||||||
|
"""Whole-word containment. Substring matching is what made this go wrong."""
|
||||||
|
return re.search(rf"(?<!\w){re.escape(needle)}(?!\w)", haystack) is not None
|
||||||
|
|
||||||
|
|
||||||
def _extract_capabilities(text: str) -> list[str]:
|
def _extract_capabilities(text: str) -> list[str]:
|
||||||
"""
|
"""
|
||||||
Extract capability names from Steward's text response.
|
Extract capability names from the Steward's declared delegation.
|
||||||
|
|
||||||
Uses keyword matching to find mentioned capabilities.
|
The prompt instructs the Steward to answer in a fixed shape::
|
||||||
|
|
||||||
|
DELEGATE: <capability> to <action> <task>
|
||||||
|
REASON: ...
|
||||||
|
COMPLEXITY: ...
|
||||||
|
CONTEXT: ...
|
||||||
|
|
||||||
|
Only the DELEGATE line states intent; the rest is free prose. An earlier
|
||||||
|
version substring-matched capability *domains* across the whole response,
|
||||||
|
which routed on ordinary English: "description" contains "script" and
|
||||||
|
"discover" contains "cover" (both housekeeper domains), "acknowledge"
|
||||||
|
contains "knowledge" and "know" (librarian, biographer), and "economy"
|
||||||
|
contains "my" (biographer). Any REASON line could therefore summon agents
|
||||||
|
the Steward never asked for, and a spurious librarian is a real
|
||||||
|
multi-second web call.
|
||||||
|
|
||||||
|
It also made prose length a routing input, so anything that shortened the
|
||||||
|
Steward's output — such as disabling model thinking — would look like it had
|
||||||
|
improved routing.
|
||||||
|
|
||||||
|
Resolution is layered, most explicit first:
|
||||||
|
1. a DELEGATE line beginning with a capability name — the documented shape
|
||||||
|
2. a capability named anywhere on a DELEGATE line
|
||||||
|
3. a capability *domain* on a DELEGATE line, for a loosely worded answer
|
||||||
|
4. no DELEGATE line: capability names only, never domains
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
text: Steward's plain text analysis
|
text: Steward's plain text analysis
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
List of capability names (e.g., ['tatlock_core'])
|
List of capability names (e.g. ['tatlock_core']), de-duplicated.
|
||||||
"""
|
"""
|
||||||
text_lower = text.lower()
|
|
||||||
registry = get_household_registry()
|
registry = get_household_registry()
|
||||||
capabilities = registry.get_all_capabilities()
|
capabilities = registry.get_all_capabilities()
|
||||||
|
delegate_lines = [line.strip().lower() for line in _DELEGATE_LINE_RE.findall(text or "")]
|
||||||
|
|
||||||
found_caps = []
|
found_caps: list[str] = []
|
||||||
|
|
||||||
for cap in capabilities:
|
def _add(name: str) -> None:
|
||||||
# Check if capability name is mentioned
|
if name not in found_caps:
|
||||||
if cap.name.lower() in text_lower:
|
found_caps.append(name)
|
||||||
found_caps.append(cap.name)
|
|
||||||
|
if not delegate_lines:
|
||||||
|
# Either the Steward judged no capability necessary — the prompt's
|
||||||
|
# conversational path, whose correct answer is [] — or it ignored the
|
||||||
|
# format. Names only: domain words are ordinary English and would fire
|
||||||
|
# on any prose, which is the bug described above.
|
||||||
|
haystack = (text or "").lower()
|
||||||
|
for cap in capabilities:
|
||||||
|
if _mentions(cap.name.lower(), haystack):
|
||||||
|
_add(cap.name)
|
||||||
|
return found_caps
|
||||||
|
|
||||||
|
for line in delegate_lines:
|
||||||
|
leading = next((c for c in capabilities if line.startswith(c.name.lower())), None)
|
||||||
|
if leading is not None:
|
||||||
|
_add(leading.name)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Check if any domains are mentioned
|
named = [c for c in capabilities if _mentions(c.name.lower(), line)]
|
||||||
for domain in cap.domains:
|
if named:
|
||||||
if domain.lower() in text_lower:
|
for cap in named:
|
||||||
found_caps.append(cap.name)
|
_add(cap.name)
|
||||||
break
|
continue
|
||||||
|
|
||||||
|
# Last resort. Scoped to this line, so the REASON and CONTEXT prose that
|
||||||
|
# caused the original misrouting can no longer reach it.
|
||||||
|
for cap in capabilities:
|
||||||
|
if any(_mentions(domain.lower(), line) for domain in cap.domains):
|
||||||
|
_add(cap.name)
|
||||||
|
|
||||||
return found_caps
|
return found_caps
|
||||||
|
|
||||||
@@ -72,8 +128,7 @@ def _extract_complexity(text: str) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def _extract_conversation_context(
|
def _extract_conversation_context(
|
||||||
text: str,
|
text: str, conversation_history: list[dict]
|
||||||
conversation_history: list[dict]
|
|
||||||
) -> ConversationContext:
|
) -> ConversationContext:
|
||||||
"""
|
"""
|
||||||
Extract conversation context analysis from text.
|
Extract conversation context analysis from text.
|
||||||
@@ -88,13 +143,15 @@ def _extract_conversation_context(
|
|||||||
text_lower = text.lower()
|
text_lower = text.lower()
|
||||||
|
|
||||||
# Check if conversation history is referenced
|
# Check if conversation history is referenced
|
||||||
has_context = bool(conversation_history) and any([
|
has_context = bool(conversation_history) and any(
|
||||||
"previous" in text_lower,
|
[
|
||||||
"earlier" in text_lower,
|
"previous" in text_lower,
|
||||||
"context" in text_lower,
|
"earlier" in text_lower,
|
||||||
"turn" in text_lower,
|
"context" in text_lower,
|
||||||
"history" in text_lower,
|
"turn" in text_lower,
|
||||||
])
|
"history" in text_lower,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
# Extract turn numbers if mentioned (e.g., "turn 0", "turn 1")
|
# Extract turn numbers if mentioned (e.g., "turn 0", "turn 1")
|
||||||
relevant_turns = []
|
relevant_turns = []
|
||||||
@@ -106,21 +163,23 @@ def _extract_conversation_context(
|
|||||||
context_summary = ""
|
context_summary = ""
|
||||||
if has_context:
|
if has_context:
|
||||||
# Extract sentence(s) mentioning context
|
# Extract sentence(s) mentioning context
|
||||||
sentences = text.split('.')
|
sentences = text.split(".")
|
||||||
context_sentences = [s for s in sentences if any(
|
context_sentences = [
|
||||||
word in s.lower() for word in ["previous", "earlier", "context", "history"]
|
s
|
||||||
)]
|
for s in sentences
|
||||||
|
if any(word in s.lower() for word in ["previous", "earlier", "context", "history"])
|
||||||
|
]
|
||||||
if context_sentences:
|
if context_sentences:
|
||||||
context_summary = context_sentences[0].strip()
|
context_summary = context_sentences[0].strip()
|
||||||
|
|
||||||
return ConversationContext(
|
return ConversationContext(
|
||||||
has_previous_context=has_context,
|
has_previous_context=has_context,
|
||||||
relevant_turns=relevant_turns,
|
relevant_turns=relevant_turns,
|
||||||
context_summary=context_summary
|
context_summary=context_summary,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _extract_missing_capabilities(text: str) -> Optional[str]:
|
def _extract_missing_capabilities(text: str) -> str | None:
|
||||||
"""
|
"""
|
||||||
Extract missing capability notes from text.
|
Extract missing capability notes from text.
|
||||||
|
|
||||||
@@ -133,24 +192,185 @@ def _extract_missing_capabilities(text: str) -> Optional[str]:
|
|||||||
text_lower = text.lower()
|
text_lower = text.lower()
|
||||||
|
|
||||||
# Look for indicators of missing capabilities
|
# Look for indicators of missing capabilities
|
||||||
if any(word in text_lower for word in [
|
if any(
|
||||||
"missing", "unavailable", "not available", "don't have", "doesn't have"
|
word in text_lower
|
||||||
]):
|
for word in ["missing", "unavailable", "not available", "don't have", "doesn't have"]
|
||||||
|
):
|
||||||
# Find the sentence mentioning missing capabilities
|
# Find the sentence mentioning missing capabilities
|
||||||
sentences = text.split('.')
|
sentences = text.split(".")
|
||||||
for sentence in sentences:
|
for sentence in sentences:
|
||||||
if any(word in sentence.lower() for word in [
|
if any(
|
||||||
"missing", "unavailable", "not available"
|
word in sentence.lower() for word in ["missing", "unavailable", "not available"]
|
||||||
]):
|
):
|
||||||
return sentence.strip()
|
return sentence.strip()
|
||||||
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _build_enriched_query(user_request: str, memory_context: dict[str, Any]) -> str:
|
||||||
|
"""
|
||||||
|
Build an enriched query by appending user context when not specified.
|
||||||
|
|
||||||
|
When the user asks location-dependent questions (weather, nearby, etc.)
|
||||||
|
without specifying a location, this appends their known location.
|
||||||
|
Similarly for timezone-dependent queries.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_request: The user's original request
|
||||||
|
memory_context: Pre-fetched memory context with profile/preferences
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: Query with context appended, or original query if no enrichment needed
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> query = _build_enriched_query(
|
||||||
|
... "What's the weather?",
|
||||||
|
... {"profile": {"location": "Amsterdam", "timezone": "Europe/Amsterdam"}}
|
||||||
|
... )
|
||||||
|
>>> query
|
||||||
|
"What's the weather?\n\n[User Context: location=Amsterdam, timezone=Europe/Amsterdam]"
|
||||||
|
"""
|
||||||
|
if not memory_context:
|
||||||
|
return user_request
|
||||||
|
|
||||||
|
request_lower = user_request.lower()
|
||||||
|
profile = memory_context.get("profile", {})
|
||||||
|
preferences = memory_context.get("preferences", {})
|
||||||
|
|
||||||
|
context_parts = []
|
||||||
|
|
||||||
|
# Check if location is needed and not specified
|
||||||
|
location_keywords = ["weather", "temperature", "forecast", "nearby", "local", "here"]
|
||||||
|
# Use word boundary pattern to avoid false positives like "at" in "what"
|
||||||
|
location_prepositions = [r"\bin\b", r"\bat\b", r"\bnear\b", r"\baround\b", r"\bfor\b"]
|
||||||
|
location_specified = any(re.search(p, request_lower) for p in location_prepositions)
|
||||||
|
|
||||||
|
if any(word in request_lower for word in location_keywords):
|
||||||
|
if not location_specified and profile.get("location"):
|
||||||
|
context_parts.append(f"location={profile['location']}")
|
||||||
|
|
||||||
|
# Check if timezone is needed and not specified
|
||||||
|
time_keywords = ["time", "schedule", "meeting", "appointment", "when", "today", "tomorrow"]
|
||||||
|
timezone_specified = any(word in request_lower for word in ["timezone", "tz", "utc", "gmt"])
|
||||||
|
|
||||||
|
if any(word in request_lower for word in time_keywords):
|
||||||
|
if not timezone_specified and profile.get("timezone"):
|
||||||
|
context_parts.append(f"timezone={profile['timezone']}")
|
||||||
|
|
||||||
|
# Add preferences if relevant
|
||||||
|
if preferences.get("temperature_unit") and "weather" in request_lower:
|
||||||
|
context_parts.append(f"temperature_unit={preferences['temperature_unit']}")
|
||||||
|
|
||||||
|
# Build enriched query
|
||||||
|
if context_parts:
|
||||||
|
context_str = ", ".join(context_parts)
|
||||||
|
return f"{user_request}\n\n[User Context: {context_str}]"
|
||||||
|
|
||||||
|
return user_request
|
||||||
|
|
||||||
|
|
||||||
|
async def _prefetch_memory_context(user_request: str) -> dict[str, Any]:
|
||||||
|
"""
|
||||||
|
Pre-fetch user context that might be needed for this request.
|
||||||
|
|
||||||
|
This is the "direct access" layer - fast lookups without LLM overhead.
|
||||||
|
Uses simple keyword matching to determine what context to fetch.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_request: The user's request text
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with profile and/or preferences data
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> ctx = await _prefetch_memory_context("What's the weather?")
|
||||||
|
>>> ctx
|
||||||
|
{"profile": {"location": "Amsterdam"}}
|
||||||
|
"""
|
||||||
|
request_lower = user_request.lower()
|
||||||
|
|
||||||
|
# Determine what context might be needed based on keywords
|
||||||
|
profile_keys = []
|
||||||
|
|
||||||
|
# Location-related queries
|
||||||
|
if any(
|
||||||
|
word in request_lower
|
||||||
|
for word in [
|
||||||
|
"weather",
|
||||||
|
"temperature",
|
||||||
|
"forecast",
|
||||||
|
"nearby",
|
||||||
|
"local",
|
||||||
|
"directions",
|
||||||
|
"distance",
|
||||||
|
"map",
|
||||||
|
"here",
|
||||||
|
# Direct location questions
|
||||||
|
"live",
|
||||||
|
"where",
|
||||||
|
"home",
|
||||||
|
"reside",
|
||||||
|
"location",
|
||||||
|
"address",
|
||||||
|
]
|
||||||
|
):
|
||||||
|
profile_keys.append("location")
|
||||||
|
|
||||||
|
# Time-related queries
|
||||||
|
if any(
|
||||||
|
word in request_lower
|
||||||
|
for word in [
|
||||||
|
"time",
|
||||||
|
"schedule",
|
||||||
|
"meeting",
|
||||||
|
"appointment",
|
||||||
|
"reminder",
|
||||||
|
"alarm",
|
||||||
|
"when",
|
||||||
|
"today",
|
||||||
|
"tomorrow",
|
||||||
|
]
|
||||||
|
):
|
||||||
|
profile_keys.append("timezone")
|
||||||
|
|
||||||
|
# Personal queries
|
||||||
|
if any(word in request_lower for word in ["my name", "who am i", "about me"]):
|
||||||
|
profile_keys.append("name")
|
||||||
|
|
||||||
|
# Always fetch preferences if they might affect response format
|
||||||
|
include_preferences = any(
|
||||||
|
word in request_lower
|
||||||
|
for word in [
|
||||||
|
"temperature",
|
||||||
|
"weather",
|
||||||
|
"convert",
|
||||||
|
"unit",
|
||||||
|
"format",
|
||||||
|
"celsius",
|
||||||
|
"fahrenheit",
|
||||||
|
"metric",
|
||||||
|
"imperial",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
return await memory_service.prefetch_context(
|
||||||
|
include_profile=bool(profile_keys),
|
||||||
|
include_preferences=include_preferences,
|
||||||
|
profile_keys=profile_keys if profile_keys else None,
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"steward_prefetch_memory_failed",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
async def analyze_request(
|
async def analyze_request(
|
||||||
user_request: str,
|
user_request: str,
|
||||||
conversation_history: list[dict],
|
conversation_history: list[dict],
|
||||||
conversation_id: Optional[str] = None,
|
conversation_id: str | None = None,
|
||||||
) -> StewardRecommendation:
|
) -> StewardRecommendation:
|
||||||
"""
|
"""
|
||||||
Analyze user request with full conversation context.
|
Analyze user request with full conversation context.
|
||||||
@@ -158,8 +378,7 @@ async def analyze_request(
|
|||||||
This is the main entry point for Steward analysis. It:
|
This is the main entry point for Steward analysis. It:
|
||||||
1. Calls the Steward agent with full conversation history
|
1. Calls the Steward agent with full conversation history
|
||||||
2. Logs the operation with timing
|
2. Logs the operation with timing
|
||||||
3. Records performance benchmarks to Redis
|
3. Returns structured recommendations
|
||||||
4. Returns structured recommendations
|
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
user_request: The current user message to analyze
|
user_request: The current user message to analyze
|
||||||
@@ -183,9 +402,13 @@ async def analyze_request(
|
|||||||
"request_preview": user_request[:100],
|
"request_preview": user_request[:100],
|
||||||
"conversation_id": conversation_id,
|
"conversation_id": conversation_id,
|
||||||
"history_length": len(conversation_history),
|
"history_length": len(conversation_history),
|
||||||
}
|
},
|
||||||
) as log_ctx:
|
) as log_ctx:
|
||||||
try:
|
try:
|
||||||
|
# Pre-fetch user context from memory (fast, no LLM)
|
||||||
|
memory_context = await _prefetch_memory_context(user_request)
|
||||||
|
log_ctx["memory_context_keys"] = list(memory_context.keys())
|
||||||
|
|
||||||
# Get Steward agent
|
# Get Steward agent
|
||||||
steward = get_steward_agent()
|
steward = get_steward_agent()
|
||||||
|
|
||||||
@@ -193,12 +416,12 @@ async def analyze_request(
|
|||||||
"steward_analyzing_request",
|
"steward_analyzing_request",
|
||||||
request=user_request,
|
request=user_request,
|
||||||
history_turns=len(conversation_history),
|
history_turns=len(conversation_history),
|
||||||
|
memory_context=bool(memory_context),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Get plain text analysis from Steward
|
# Get plain text analysis from Steward
|
||||||
analysis_text = await steward.analyze(
|
analysis_text = await steward.analyze(
|
||||||
user_request,
|
user_request, conversation_history=conversation_history
|
||||||
conversation_history=conversation_history
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Parse plain text into structured recommendation
|
# Parse plain text into structured recommendation
|
||||||
@@ -207,12 +430,17 @@ async def analyze_request(
|
|||||||
context = _extract_conversation_context(analysis_text, conversation_history)
|
context = _extract_conversation_context(analysis_text, conversation_history)
|
||||||
missing = _extract_missing_capabilities(analysis_text)
|
missing = _extract_missing_capabilities(analysis_text)
|
||||||
|
|
||||||
|
# Build enriched query with auto-filled context
|
||||||
|
enriched_query = _build_enriched_query(user_request, memory_context)
|
||||||
|
|
||||||
recommendation = StewardRecommendation(
|
recommendation = StewardRecommendation(
|
||||||
recommended_capabilities=capabilities,
|
recommended_capabilities=capabilities,
|
||||||
reasoning=analysis_text,
|
reasoning=analysis_text,
|
||||||
estimated_complexity=complexity,
|
estimated_complexity=complexity,
|
||||||
conversation_context=context,
|
conversation_context=context,
|
||||||
missing_capabilities=missing
|
missing_capabilities=missing,
|
||||||
|
memory_context=memory_context,
|
||||||
|
enriched_query=enriched_query,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Update log context with results
|
# Update log context with results
|
||||||
@@ -228,23 +456,6 @@ async def analyze_request(
|
|||||||
reasoning=analysis_text[:200], # First 200 chars
|
reasoning=analysis_text[:200], # First 200 chars
|
||||||
)
|
)
|
||||||
|
|
||||||
# Record performance benchmark
|
|
||||||
if log_ctx.get("duration_seconds"):
|
|
||||||
benchmark = PerformanceBenchmark(
|
|
||||||
operation="steward_analysis",
|
|
||||||
duration_seconds=log_ctx["duration_seconds"],
|
|
||||||
success=True,
|
|
||||||
recommendation_count=len(recommendation.recommended_capabilities),
|
|
||||||
confidence=None, # Could add confidence scoring in future
|
|
||||||
conversation_id=conversation_id,
|
|
||||||
metadata={
|
|
||||||
"complexity": recommendation.estimated_complexity,
|
|
||||||
"has_context": recommendation.conversation_context.has_previous_context,
|
|
||||||
"missing_capabilities": recommendation.missing_capabilities is not None,
|
|
||||||
},
|
|
||||||
)
|
|
||||||
await get_benchmark_store().record(benchmark)
|
|
||||||
|
|
||||||
return recommendation
|
return recommendation
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
+458
-144
@@ -6,21 +6,26 @@ The agent embodies a witty, capable British butler personality.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import secrets
|
import secrets
|
||||||
from typing import AsyncGenerator, Any
|
from collections.abc import AsyncGenerator
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from pydantic_ai import Agent, RunContext
|
from pydantic_ai import Agent, RunContext
|
||||||
|
|
||||||
from src.agents.base import AgentInterface, OutputItem
|
from src.agents.base import AgentInterface, OutputItem
|
||||||
from src.agents.tatlock_core.tools import (
|
from src.agents.tatlock_core.tools import (
|
||||||
calculate,
|
calculate,
|
||||||
get_current_datetime,
|
|
||||||
calculate_time_offset,
|
calculate_time_offset,
|
||||||
|
get_current_datetime,
|
||||||
time_difference,
|
time_difference,
|
||||||
search_web,
|
|
||||||
)
|
)
|
||||||
from src.core.config import config
|
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
|
from src.core.tracing import (
|
||||||
|
SpanType,
|
||||||
|
add_tool_spans_from_messages,
|
||||||
|
end_span,
|
||||||
|
start_span,
|
||||||
|
)
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
@@ -28,9 +33,10 @@ logger = get_logger(__name__)
|
|||||||
@dataclass
|
@dataclass
|
||||||
class ToolCallTracker:
|
class ToolCallTracker:
|
||||||
"""Tracks tool calls for reporting to reasoning output."""
|
"""Tracks tool calls for reporting to reasoning output."""
|
||||||
|
|
||||||
calls: list[str] = field(default_factory=list)
|
calls: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
def log_call(self, message: str):
|
def log_call(self, message: str) -> None:
|
||||||
"""Log a tool call."""
|
"""Log a tool call."""
|
||||||
self.calls.append(message)
|
self.calls.append(message)
|
||||||
|
|
||||||
@@ -43,7 +49,16 @@ def generate_id() -> str:
|
|||||||
# System prompt defining Tatlock's personality
|
# System prompt defining Tatlock's personality
|
||||||
TATLOCK_SYSTEM_PROMPT = """You are Tatlock, a helpful personal assistant with the demeanor of a British butler.
|
TATLOCK_SYSTEM_PROMPT = """You are Tatlock, a helpful personal assistant with the demeanor of a British butler.
|
||||||
|
|
||||||
Address users as "sir" and maintain a formal yet personable tone. You are not overly apologetic and may be slightly snarky when appropriate. If an opportunity for a pun presents itself, you cannot resist.
|
## Personality
|
||||||
|
|
||||||
|
Address users as "sir". Be confident, direct, and efficient - you are an unflappable English butler who gets things done. Dry wit and puns are encouraged.
|
||||||
|
|
||||||
|
**CRITICAL - Do NOT:**
|
||||||
|
- Apologize unless you genuinely made an error
|
||||||
|
- Say "Apologies for any confusion" or "Allow me to rectify" when nothing went wrong
|
||||||
|
- Preface successful results with caveats or apologies
|
||||||
|
|
||||||
|
When presenting findings: lead with the answer, be concise, skip the preamble.
|
||||||
|
|
||||||
You coordinate with various household staff (expert agents) to provide comprehensive assistance across:
|
You coordinate with various household staff (expert agents) to provide comprehensive assistance across:
|
||||||
- Research and knowledge work
|
- Research and knowledge work
|
||||||
@@ -76,25 +91,62 @@ You have direct access to several permanent tools that you should USE whenever a
|
|||||||
- time_difference: Calculate the time between two dates
|
- time_difference: Calculate the time between two dates
|
||||||
- Use these for ANY date/time queries - never guess at dates or times
|
- Use these for ANY date/time queries - never guess at dates or times
|
||||||
|
|
||||||
3. **Web Search** (search_web): Search for current, volatile, or factual information
|
3. **Web Search** (via Librarian): For current, volatile, or factual information
|
||||||
- Use this for ANY information that might be current, factual, or outside your training data
|
- Delegate to the Librarian for web searches and research
|
||||||
- Examples: news, current events, recent developments, specific facts, technical documentation
|
- Examples: news, current events, recent developments, specific facts, technical documentation
|
||||||
- Always prefer searching over guessing or using potentially outdated knowledge
|
- Use: delegate_to_librarian(task="search the web for ...")
|
||||||
- For extensive research questions, note that this will later be delegated to the librarian
|
|
||||||
|
|
||||||
## Tool Usage Guidelines
|
## Tool Usage Guidelines
|
||||||
|
|
||||||
- **Mathematics**: ALWAYS use the calculator tool, even for simple arithmetic
|
- **Mathematics**: ALWAYS use the calculator tool, even for simple arithmetic
|
||||||
- **Dates/Times**: ALWAYS use the date/time tools, never guess or estimate
|
- **Dates/Times**: ALWAYS use the date/time tools, never guess or estimate
|
||||||
- **Current Information**: ALWAYS search for facts, news, or volatile information
|
- **Current Information**: Delegate web searches to the Librarian
|
||||||
- **Verification**: When facts are important, use search to verify rather than rely on memory alone
|
- **Verification**: When facts are important, delegate to Librarian for research
|
||||||
- When you use a tool, explain what you're doing in a butler-appropriate manner
|
- When you use a tool, explain what you're doing in a butler-appropriate manner
|
||||||
- Present tool results naturally in your response
|
- Present tool results naturally in your response
|
||||||
|
|
||||||
Currently in Phase 1 development - expert agent delegation will be added in later phases.
|
## Expert Delegation (CRITICAL)
|
||||||
|
|
||||||
|
When you see "DELEGATE:" in your instructions, you MUST delegate to the appropriate agent.
|
||||||
|
|
||||||
|
**PRIMARY METHOD**: Call the delegation function directly:
|
||||||
|
- `delegate_to_librarian(task="...")` for research/wiki tasks
|
||||||
|
- `delegate_to_biographer(task="...")` for memory tasks
|
||||||
|
|
||||||
|
**FALLBACK METHOD**: If function calling fails, output EXACTLY this format:
|
||||||
|
```
|
||||||
|
[DELEGATE:biographer] task="Remember that user's name is TestBot"
|
||||||
|
```
|
||||||
|
or
|
||||||
|
```
|
||||||
|
[DELEGATE:librarian] task="Search for information about Docker"
|
||||||
|
```
|
||||||
|
|
||||||
|
**Rules:**
|
||||||
|
1. When you see "DELEGATE: biographer" - delegate to biographer
|
||||||
|
2. When you see "DELEGATE: librarian" - delegate to librarian
|
||||||
|
3. NEVER ask for confirmation - just delegate
|
||||||
|
4. NEVER handle delegated tasks yourself
|
||||||
|
5. If you cannot call the function, use the [DELEGATE:...] text format EXACTLY
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
# Tool-phase prompt for orchestrate_tool_calls(). The butler personality prompt
|
||||||
|
# suppresses tool calling on small local models (gemma4 reasons about the tool,
|
||||||
|
# then answers from memory with wrong arithmetic), so the orchestration phase
|
||||||
|
# uses a terse operator prompt; synthesize_from_results() applies the persona.
|
||||||
|
TATLOCK_ORCHESTRATION_PROMPT = """You are the tool-execution phase of Tatlock, \
|
||||||
|
a butler assistant. Your only job is to gather accurate results by calling the \
|
||||||
|
provided tools.
|
||||||
|
|
||||||
|
- ALWAYS use tools for the task - never answer from memory and never do mental math.
|
||||||
|
- Mathematics: call the calculate tool, even for trivial arithmetic.
|
||||||
|
- Dates and times: call the date/time tools, never guess.
|
||||||
|
- When the instructions say DELEGATE to an agent, call the matching delegate_to_* tool.
|
||||||
|
- After the tool results arrive, reply with a one-line factual summary of the results. \
|
||||||
|
A later step writes the polished reply, so do not add personality."""
|
||||||
|
|
||||||
|
|
||||||
class TatlockAgent(AgentInterface):
|
class TatlockAgent(AgentInterface):
|
||||||
"""
|
"""
|
||||||
Tatlock - The Butler agent using PydanticAI with Ollama.
|
Tatlock - The Butler agent using PydanticAI with Ollama.
|
||||||
@@ -103,50 +155,47 @@ class TatlockAgent(AgentInterface):
|
|||||||
currently in Phase 1 (basic LLM integration without expert agents).
|
currently in Phase 1 (basic LLM integration without expert agents).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self):
|
def __init__(self) -> None:
|
||||||
"""Initialize Tatlock configuration (lazy agent creation)."""
|
"""Initialize Tatlock (lazy agent creation)."""
|
||||||
# Store Ollama configuration
|
# Deps are a ToolCallTracker: every registered tool takes
|
||||||
self.ollama_host = str(config.OLLAMA_HOST)
|
# RunContext[ToolCallTracker], and run() is called with one. Saying so
|
||||||
self.model_name = config.OLLAMA_DEFAULT_MODEL
|
# is what lets the tool registrations below type-check at all.
|
||||||
self._agent = None # Lazy initialization
|
self._agent: Agent[ToolCallTracker, str] | None = None # Lazy initialization
|
||||||
|
|
||||||
def _ensure_agent(self):
|
def _ensure_agent(self) -> None:
|
||||||
"""Ensure the PydanticAI agent is initialized (lazy initialization)."""
|
"""Ensure the PydanticAI agent is initialized (lazy initialization)."""
|
||||||
if self._agent is not None:
|
if self._agent is not None:
|
||||||
return
|
return
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model, get_model_info
|
||||||
|
|
||||||
|
model_info = get_model_info()
|
||||||
logger.info(
|
logger.info(
|
||||||
"tatlock_agent_initializing",
|
"tatlock_agent_initializing",
|
||||||
ollama_host=self.ollama_host,
|
backend=model_info["backend"],
|
||||||
model=self.model_name,
|
model=model_info["model"],
|
||||||
)
|
)
|
||||||
|
|
||||||
# Import required classes for Ollama configuration
|
# Get best available model (Claude if available, else Ollama)
|
||||||
from pydantic_ai.models.openai import OpenAIChatModel
|
model = get_model()
|
||||||
from pydantic_ai.providers.ollama import OllamaProvider
|
|
||||||
|
|
||||||
# PydanticAI expects Ollama base URL to end with /v1
|
# Create PydanticAI agent
|
||||||
# Remove trailing slash from ollama_host if present
|
|
||||||
clean_host = self.ollama_host.rstrip('/')
|
|
||||||
base_url = f"{clean_host}/v1"
|
|
||||||
|
|
||||||
# Create Ollama model with provider
|
|
||||||
ollama_model = OpenAIChatModel(
|
|
||||||
model_name=self.model_name,
|
|
||||||
provider=OllamaProvider(base_url=base_url)
|
|
||||||
)
|
|
||||||
|
|
||||||
# Create PydanticAI agent with Ollama model
|
|
||||||
self._agent = Agent(
|
self._agent = Agent(
|
||||||
ollama_model,
|
model,
|
||||||
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
||||||
|
deps_type=ToolCallTracker,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Register tools with the agent
|
# Register tools with the agent
|
||||||
self._register_tools()
|
self._register_tools()
|
||||||
|
|
||||||
def _register_tools(self):
|
def _register_tools(self) -> None:
|
||||||
"""Register permanent tools with the PydanticAI agent."""
|
"""Register permanent tools with the PydanticAI agent.
|
||||||
|
|
||||||
|
Called only from _ensure_agent, immediately after the agent is built, so
|
||||||
|
the assert documents an invariant rather than guarding a real case.
|
||||||
|
"""
|
||||||
|
assert self._agent is not None, "_register_tools called before the agent exists"
|
||||||
|
|
||||||
# Calculator tool
|
# Calculator tool
|
||||||
@self._agent.tool
|
@self._agent.tool
|
||||||
@@ -201,7 +250,9 @@ class TatlockAgent(AgentInterface):
|
|||||||
|
|
||||||
# Time difference calculator
|
# Time difference calculator
|
||||||
@self._agent.tool
|
@self._agent.tool
|
||||||
def calculate_time_difference(ctx: RunContext[ToolCallTracker], date1_str: str, date2_str: str = "now") -> str:
|
def calculate_time_difference(
|
||||||
|
ctx: RunContext[ToolCallTracker], date1_str: str, date2_str: str = "now"
|
||||||
|
) -> str:
|
||||||
"""
|
"""
|
||||||
Calculate the difference between two dates.
|
Calculate the difference between two dates.
|
||||||
|
|
||||||
@@ -213,31 +264,13 @@ class TatlockAgent(AgentInterface):
|
|||||||
Human-readable description of the time difference
|
Human-readable description of the time difference
|
||||||
"""
|
"""
|
||||||
if ctx.deps:
|
if ctx.deps:
|
||||||
ctx.deps.log_call(f"🕐 Calculating time difference between {date1_str} and {date2_str}")
|
ctx.deps.log_call(
|
||||||
|
f"🕐 Calculating time difference between {date1_str} and {date2_str}"
|
||||||
|
)
|
||||||
return time_difference(date1_str, date2_str)
|
return time_difference(date1_str, date2_str)
|
||||||
|
|
||||||
# Web search tool
|
# NOTE: Web search has been moved to The Librarian agent.
|
||||||
@self._agent.tool
|
# Use delegate_to_librarian(task="search web for ...") for web search.
|
||||||
async def web_search(ctx: RunContext[ToolCallTracker], query: str, num_results: int = 5) -> str:
|
|
||||||
"""
|
|
||||||
Search the web using SearXNG for current information.
|
|
||||||
|
|
||||||
Use this tool for ANY information that might be:
|
|
||||||
- Current or time-sensitive (news, events, recent developments)
|
|
||||||
- Factual and verifiable (statistics, technical specs, definitions)
|
|
||||||
- Outside your training data or knowledge cutoff
|
|
||||||
|
|
||||||
Args:
|
|
||||||
query: Search query string
|
|
||||||
num_results: Number of results to return (default: 5, max: 10)
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Formatted search results with titles, URLs, and snippets
|
|
||||||
"""
|
|
||||||
# Log the search query to reasoning output
|
|
||||||
if ctx.deps:
|
|
||||||
ctx.deps.log_call(f"🔍 Searching for: '{query}'")
|
|
||||||
return await search_web(query, num_results)
|
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def agent(self):
|
def agent(self):
|
||||||
@@ -253,7 +286,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
temperature: float = 1.0,
|
temperature: float = 1.0,
|
||||||
max_tokens: int | None = None,
|
max_tokens: int | None = None,
|
||||||
stop: list[str] | None = None,
|
stop: list[str] | None = None,
|
||||||
**kwargs: Any
|
**kwargs: Any,
|
||||||
) -> AsyncGenerator[OutputItem, None]:
|
) -> AsyncGenerator[OutputItem, None]:
|
||||||
"""
|
"""
|
||||||
Generate response using PydanticAI with Ollama.
|
Generate response using PydanticAI with Ollama.
|
||||||
@@ -287,20 +320,28 @@ class TatlockAgent(AgentInterface):
|
|||||||
type="message",
|
type="message",
|
||||||
id=f"msg_{generate_id()}",
|
id=f"msg_{generate_id()}",
|
||||||
role="assistant",
|
role="assistant",
|
||||||
content=[{
|
content=[
|
||||||
"type": "output_text",
|
{
|
||||||
"text": "I'm afraid I didn't receive a message, sir. How may I assist you?",
|
"type": "output_text",
|
||||||
"annotations": []
|
"text": "I'm afraid I didn't receive a message, sir. How may I assist you?",
|
||||||
}],
|
"annotations": [],
|
||||||
status="completed"
|
}
|
||||||
|
],
|
||||||
|
status="completed",
|
||||||
)
|
)
|
||||||
return
|
return
|
||||||
|
|
||||||
# Build message history (all messages except the last user message)
|
# Build message history (all messages except the last user message)
|
||||||
# PydanticAI expects history as list of ModelRequest/ModelResponse objects
|
# PydanticAI expects history as list of ModelRequest/ModelResponse objects
|
||||||
from pydantic_ai.messages import ModelRequest, ModelResponse, UserPromptPart, TextPart
|
from pydantic_ai.messages import (
|
||||||
|
ModelMessage,
|
||||||
|
ModelRequest,
|
||||||
|
ModelResponse,
|
||||||
|
TextPart,
|
||||||
|
UserPromptPart,
|
||||||
|
)
|
||||||
|
|
||||||
message_history = []
|
message_history: list[ModelMessage] = []
|
||||||
for i, msg in enumerate(messages[:-1]): # All messages except the last one
|
for i, msg in enumerate(messages[:-1]): # All messages except the last one
|
||||||
role = msg.get("role")
|
role = msg.get("role")
|
||||||
content = msg.get("content", "")
|
content = msg.get("content", "")
|
||||||
@@ -312,7 +353,9 @@ class TatlockAgent(AgentInterface):
|
|||||||
|
|
||||||
# Debug: Check for problematic content
|
# Debug: Check for problematic content
|
||||||
if '"' in content or "'" in content:
|
if '"' in content or "'" in content:
|
||||||
logger.debug(f"Message {i} ({role}) contains quotes. Content preview: {content[:100]}...")
|
logger.debug(
|
||||||
|
f"Message {i} ({role}) contains quotes. Content preview: {content[:100]}..."
|
||||||
|
)
|
||||||
|
|
||||||
# Convert to PydanticAI message format
|
# Convert to PydanticAI message format
|
||||||
try:
|
try:
|
||||||
@@ -321,9 +364,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
ModelRequest(parts=[UserPromptPart(content=content)])
|
ModelRequest(parts=[UserPromptPart(content=content)])
|
||||||
)
|
)
|
||||||
elif role == "assistant":
|
elif role == "assistant":
|
||||||
message_history.append(
|
message_history.append(ModelResponse(parts=[TextPart(content=content)]))
|
||||||
ModelResponse(parts=[TextPart(content=content)])
|
|
||||||
)
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"Error creating message history item {i}: {e}")
|
logger.error(f"Error creating message history item {i}: {e}")
|
||||||
logger.error(f"Problematic content: {repr(content)}")
|
logger.error(f"Problematic content: {repr(content)}")
|
||||||
@@ -334,7 +375,9 @@ class TatlockAgent(AgentInterface):
|
|||||||
if message_history:
|
if message_history:
|
||||||
for i, hist_msg in enumerate(message_history):
|
for i, hist_msg in enumerate(message_history):
|
||||||
msg_type = type(hist_msg).__name__
|
msg_type = type(hist_msg).__name__
|
||||||
content_preview = str(hist_msg.parts[0].content)[:50] if hist_msg.parts else "no parts"
|
content_preview = (
|
||||||
|
str(hist_msg.parts[0].content)[:50] if hist_msg.parts else "no parts"
|
||||||
|
)
|
||||||
logger.info(f" History[{i}]: {msg_type} - {content_preview}...")
|
logger.info(f" History[{i}]: {msg_type} - {content_preview}...")
|
||||||
|
|
||||||
# Generate reasoning output if requested
|
# Generate reasoning output if requested
|
||||||
@@ -344,10 +387,10 @@ class TatlockAgent(AgentInterface):
|
|||||||
id=f"reasoning_{generate_id()}",
|
id=f"reasoning_{generate_id()}",
|
||||||
summary=[
|
summary=[
|
||||||
"Analyzing your request, sir...",
|
"Analyzing your request, sir...",
|
||||||
"Formulating response based on available knowledge..."
|
"Formulating response based on available knowledge...",
|
||||||
],
|
],
|
||||||
thinking="", # PydanticAI doesn't expose internal reasoning yet
|
thinking="", # PydanticAI doesn't expose internal reasoning yet
|
||||||
status="completed"
|
status="completed",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create a tool call tracker for this request
|
# Create a tool call tracker for this request
|
||||||
@@ -364,7 +407,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
result = await self.agent.run(
|
result = await self.agent.run(
|
||||||
user_message,
|
user_message,
|
||||||
message_history=message_history if message_history else None,
|
message_history=message_history if message_history else None,
|
||||||
deps=tracker
|
deps=tracker,
|
||||||
)
|
)
|
||||||
final_text = result.output
|
final_text = result.output
|
||||||
|
|
||||||
@@ -375,7 +418,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
id=f"reasoning_tools_{generate_id()}",
|
id=f"reasoning_tools_{generate_id()}",
|
||||||
summary=tracker.calls,
|
summary=tracker.calls,
|
||||||
thinking="",
|
thinking="",
|
||||||
status="completed"
|
status="completed",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Yield the complete message
|
# Yield the complete message
|
||||||
@@ -384,12 +427,8 @@ class TatlockAgent(AgentInterface):
|
|||||||
type="message",
|
type="message",
|
||||||
id=msg_id,
|
id=msg_id,
|
||||||
role="assistant",
|
role="assistant",
|
||||||
content=[{
|
content=[{"type": "output_text", "text": final_text, "annotations": []}],
|
||||||
"type": "output_text",
|
status="completed",
|
||||||
"text": final_text,
|
|
||||||
"annotations": []
|
|
||||||
}],
|
|
||||||
status="completed"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -398,12 +437,14 @@ class TatlockAgent(AgentInterface):
|
|||||||
type="message",
|
type="message",
|
||||||
id=f"msg_{generate_id()}",
|
id=f"msg_{generate_id()}",
|
||||||
role="assistant",
|
role="assistant",
|
||||||
content=[{
|
content=[
|
||||||
"type": "output_text",
|
{
|
||||||
"text": f"My apologies, sir. I encountered an error: {str(e)}",
|
"type": "output_text",
|
||||||
"annotations": []
|
"text": f"My apologies, sir. I encountered an error: {str(e)}",
|
||||||
}],
|
"annotations": [],
|
||||||
status="failed"
|
}
|
||||||
|
],
|
||||||
|
status="failed",
|
||||||
)
|
)
|
||||||
|
|
||||||
async def supports_tools(self) -> bool:
|
async def supports_tools(self) -> bool:
|
||||||
@@ -433,7 +474,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
steward_note: Note from Steward (prepended to request, invisible to user)
|
steward_note: Note from Steward (prepended to request, invisible to user)
|
||||||
scoped_tools: List of tool definitions from household registry
|
scoped_tools: List of tool definitions from household registry
|
||||||
message_history: Conversation history in PydanticAI format
|
message_history: Conversation history in PydanticAI format
|
||||||
tool_tracker: Optional tool call tracker for benchmarking
|
tool_tracker: Optional tool call tracker for analysis
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
str: Tatlock's response text
|
str: Tatlock's response text
|
||||||
@@ -447,8 +488,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
... tool_tracker=tracker,
|
... tool_tracker=tracker,
|
||||||
... )
|
... )
|
||||||
"""
|
"""
|
||||||
from pydantic_ai.models.openai import OpenAIChatModel
|
from src.anthropic.model_selector import get_model
|
||||||
from pydantic_ai.providers.ollama import OllamaProvider
|
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"tatlock_run_with_scoped_tools",
|
"tatlock_run_with_scoped_tools",
|
||||||
@@ -459,18 +499,12 @@ class TatlockAgent(AgentInterface):
|
|||||||
|
|
||||||
# Create a fresh agent instance with scoped tools only
|
# Create a fresh agent instance with scoped tools only
|
||||||
# This ensures Tatlock can ONLY use tools recommended by the Steward
|
# This ensures Tatlock can ONLY use tools recommended by the Steward
|
||||||
clean_host = self.ollama_host.rstrip('/')
|
model = get_model()
|
||||||
base_url = f"{clean_host}/v1"
|
|
||||||
|
|
||||||
ollama_model = OpenAIChatModel(
|
|
||||||
model_name=self.model_name,
|
|
||||||
provider=OllamaProvider(base_url=base_url)
|
|
||||||
)
|
|
||||||
|
|
||||||
# Create agent with scoped tools
|
# Create agent with scoped tools
|
||||||
# Tools from household registry are already PydanticAI Tool objects
|
# Tools from household registry are already PydanticAI Tool objects
|
||||||
scoped_agent = Agent(
|
scoped_agent = Agent(
|
||||||
ollama_model,
|
model,
|
||||||
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
||||||
tools=scoped_tools, # Pass tools directly to Agent constructor
|
tools=scoped_tools, # Pass tools directly to Agent constructor
|
||||||
)
|
)
|
||||||
@@ -479,9 +513,15 @@ class TatlockAgent(AgentInterface):
|
|||||||
enriched_message = f"{steward_note}\n\n{user_message}"
|
enriched_message = f"{steward_note}\n\n{user_message}"
|
||||||
|
|
||||||
# Convert message history to PydanticAI format
|
# Convert message history to PydanticAI format
|
||||||
from pydantic_ai.messages import ModelRequest, ModelResponse, UserPromptPart, TextPart
|
from pydantic_ai.messages import (
|
||||||
|
ModelMessage,
|
||||||
|
ModelRequest,
|
||||||
|
ModelResponse,
|
||||||
|
TextPart,
|
||||||
|
UserPromptPart,
|
||||||
|
)
|
||||||
|
|
||||||
pydantic_history = []
|
pydantic_history: list[ModelMessage] = []
|
||||||
for msg in message_history:
|
for msg in message_history:
|
||||||
role = msg.get("role")
|
role = msg.get("role")
|
||||||
content = msg.get("content", "")
|
content = msg.get("content", "")
|
||||||
@@ -490,19 +530,19 @@ class TatlockAgent(AgentInterface):
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
if role == "user":
|
if role == "user":
|
||||||
pydantic_history.append(
|
pydantic_history.append(ModelRequest(parts=[UserPromptPart(content=content)]))
|
||||||
ModelRequest(parts=[UserPromptPart(content=content)])
|
|
||||||
)
|
|
||||||
elif role == "assistant":
|
elif role == "assistant":
|
||||||
pydantic_history.append(
|
pydantic_history.append(ModelResponse(parts=[TextPart(content=content)]))
|
||||||
ModelResponse(parts=[TextPart(content=content)])
|
|
||||||
)
|
|
||||||
|
|
||||||
# Run with scoped tools and tracker
|
# Run with scoped tools and tracker
|
||||||
|
# Force tool_choice to make LLM actually call tools
|
||||||
|
from src.anthropic.model_selector import get_tool_choice_settings
|
||||||
|
|
||||||
result = await scoped_agent.run(
|
result = await scoped_agent.run(
|
||||||
enriched_message,
|
enriched_message,
|
||||||
message_history=pydantic_history if pydantic_history else None,
|
message_history=pydantic_history if pydantic_history else None,
|
||||||
deps=tool_tracker
|
deps=tool_tracker,
|
||||||
|
model_settings=get_tool_choice_settings(),
|
||||||
)
|
)
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
@@ -538,8 +578,7 @@ class TatlockAgent(AgentInterface):
|
|||||||
Yields:
|
Yields:
|
||||||
Text chunks from the streaming response
|
Text chunks from the streaming response
|
||||||
"""
|
"""
|
||||||
from pydantic_ai.models.openai import OpenAIChatModel
|
from src.anthropic.model_selector import get_model
|
||||||
from pydantic_ai.providers.ollama import OllamaProvider
|
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"tatlock_run_with_scoped_tools_stream",
|
"tatlock_run_with_scoped_tools_stream",
|
||||||
@@ -549,17 +588,11 @@ class TatlockAgent(AgentInterface):
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Create a fresh agent instance with scoped tools only
|
# Create a fresh agent instance with scoped tools only
|
||||||
clean_host = self.ollama_host.rstrip('/')
|
model = get_model()
|
||||||
base_url = f"{clean_host}/v1"
|
|
||||||
|
|
||||||
ollama_model = OpenAIChatModel(
|
|
||||||
model_name=self.model_name,
|
|
||||||
provider=OllamaProvider(base_url=base_url)
|
|
||||||
)
|
|
||||||
|
|
||||||
# Create agent with scoped tools
|
# Create agent with scoped tools
|
||||||
scoped_agent = Agent(
|
scoped_agent = Agent(
|
||||||
ollama_model,
|
model,
|
||||||
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
||||||
tools=scoped_tools,
|
tools=scoped_tools,
|
||||||
)
|
)
|
||||||
@@ -568,9 +601,15 @@ class TatlockAgent(AgentInterface):
|
|||||||
enriched_message = f"{steward_note}\n\n{user_message}"
|
enriched_message = f"{steward_note}\n\n{user_message}"
|
||||||
|
|
||||||
# Convert message history to PydanticAI format
|
# Convert message history to PydanticAI format
|
||||||
from pydantic_ai.messages import ModelRequest, ModelResponse, UserPromptPart, TextPart
|
from pydantic_ai.messages import (
|
||||||
|
ModelMessage,
|
||||||
|
ModelRequest,
|
||||||
|
ModelResponse,
|
||||||
|
TextPart,
|
||||||
|
UserPromptPart,
|
||||||
|
)
|
||||||
|
|
||||||
pydantic_history = []
|
pydantic_history: list[ModelMessage] = []
|
||||||
for msg in message_history:
|
for msg in message_history:
|
||||||
role = msg.get("role")
|
role = msg.get("role")
|
||||||
content = msg.get("content", "")
|
content = msg.get("content", "")
|
||||||
@@ -579,31 +618,306 @@ class TatlockAgent(AgentInterface):
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
if role == "user":
|
if role == "user":
|
||||||
pydantic_history.append(
|
pydantic_history.append(ModelRequest(parts=[UserPromptPart(content=content)]))
|
||||||
ModelRequest(parts=[UserPromptPart(content=content)])
|
|
||||||
)
|
|
||||||
elif role == "assistant":
|
elif role == "assistant":
|
||||||
pydantic_history.append(
|
pydantic_history.append(ModelResponse(parts=[TextPart(content=content)]))
|
||||||
ModelResponse(parts=[TextPart(content=content)])
|
|
||||||
)
|
|
||||||
|
|
||||||
# Stream with scoped tools and tracker
|
# Use run() instead of run_stream() to avoid Ollama 400 bug
|
||||||
async with scoped_agent.run_stream(
|
# with streaming + tool calls (PydanticAI issues #1292, #2256)
|
||||||
|
# We yield the final response in chunks to maintain streaming interface
|
||||||
|
result = await scoped_agent.run(
|
||||||
enriched_message,
|
enriched_message,
|
||||||
message_history=pydantic_history if pydantic_history else None,
|
message_history=pydantic_history if pydantic_history else None,
|
||||||
deps=tool_tracker
|
deps=tool_tracker,
|
||||||
) as stream:
|
)
|
||||||
async for chunk in stream.stream_text(delta=True):
|
|
||||||
yield chunk
|
|
||||||
|
|
||||||
logger.info("tatlock_stream_complete")
|
# Stream the final response in chunks to maintain UX
|
||||||
|
response_text = result.output
|
||||||
|
chunk_size = 50 # characters per chunk
|
||||||
|
|
||||||
|
for i in range(0, len(response_text), chunk_size):
|
||||||
|
yield response_text[i : i + chunk_size]
|
||||||
|
|
||||||
|
logger.info("tatlock_scoped_run_complete")
|
||||||
|
|
||||||
|
async def orchestrate_tool_calls(
|
||||||
|
self,
|
||||||
|
user_message: str,
|
||||||
|
steward_note: str,
|
||||||
|
scoped_tools: list[Any],
|
||||||
|
message_history: list[dict],
|
||||||
|
tool_tracker: Any = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""
|
||||||
|
Phase 1: Execute tool calls and delegations, return structured results.
|
||||||
|
|
||||||
|
This is the coordination phase where Tatlock orchestrates tool calls
|
||||||
|
and expert delegations. The raw output is captured for Phase 2 synthesis.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_message: The user's original message
|
||||||
|
steward_note: Note from Steward (invisible to user)
|
||||||
|
scoped_tools: List of tool definitions from household registry
|
||||||
|
message_history: Conversation history
|
||||||
|
tool_tracker: Optional tool call tracker for analysis
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict with:
|
||||||
|
- tools_called: List of tool names that were called
|
||||||
|
- expert_results: Dict mapping expert names to their outputs
|
||||||
|
- tool_outputs: Dict mapping tool names to their outputs
|
||||||
|
- raw_output: The agent's raw text output
|
||||||
|
"""
|
||||||
|
from pydantic_ai.messages import (
|
||||||
|
ModelMessage,
|
||||||
|
ModelRequest,
|
||||||
|
ModelResponse,
|
||||||
|
TextPart,
|
||||||
|
ToolCallPart,
|
||||||
|
ToolReturnPart,
|
||||||
|
UserPromptPart,
|
||||||
|
)
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"tatlock_orchestrate_tool_calls",
|
||||||
|
user_message_preview=user_message[:100],
|
||||||
|
scoped_tool_count=len(scoped_tools),
|
||||||
|
history_length=len(message_history),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Start tracing span for orchestration phase
|
||||||
|
orchestrate_span = start_span(
|
||||||
|
"tatlock_orchestrate",
|
||||||
|
SpanType.TATLOCK,
|
||||||
|
metadata={
|
||||||
|
"scoped_tool_count": len(scoped_tools),
|
||||||
|
"tool_names": [getattr(t, "__name__", str(t)) for t in scoped_tools[:5]],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a fresh agent instance with scoped tools only
|
||||||
|
model = get_model()
|
||||||
|
|
||||||
|
# Create agent with scoped tools, using the tool-phase prompt
|
||||||
|
scoped_agent = Agent(
|
||||||
|
model,
|
||||||
|
system_prompt=TATLOCK_ORCHESTRATION_PROMPT,
|
||||||
|
tools=scoped_tools,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Prepend Steward's note to the request
|
||||||
|
enriched_message = f"{steward_note}\n\n{user_message}"
|
||||||
|
|
||||||
|
# Convert message history to PydanticAI format
|
||||||
|
pydantic_history: list[ModelMessage] = []
|
||||||
|
for msg in message_history:
|
||||||
|
role = msg.get("role")
|
||||||
|
content = msg.get("content", "")
|
||||||
|
|
||||||
|
if not content or not content.strip():
|
||||||
|
continue
|
||||||
|
|
||||||
|
if role == "user":
|
||||||
|
pydantic_history.append(ModelRequest(parts=[UserPromptPart(content=content)]))
|
||||||
|
elif role == "assistant":
|
||||||
|
pydantic_history.append(ModelResponse(parts=[TextPart(content=content)]))
|
||||||
|
|
||||||
|
# Run with scoped tools and tracker
|
||||||
|
from src.anthropic.model_selector import get_tool_choice_settings
|
||||||
|
|
||||||
|
result = await scoped_agent.run(
|
||||||
|
enriched_message,
|
||||||
|
message_history=pydantic_history if pydantic_history else None,
|
||||||
|
deps=tool_tracker,
|
||||||
|
model_settings=get_tool_choice_settings(),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Extract tool calls and results from the agent's messages
|
||||||
|
tools_called = []
|
||||||
|
expert_results: dict[str, Any] = {}
|
||||||
|
tool_outputs: dict[str, Any] = {}
|
||||||
|
|
||||||
|
# Parse through new messages to find tool calls and returns
|
||||||
|
for msg in result.new_messages():
|
||||||
|
if isinstance(msg, ModelResponse):
|
||||||
|
for part in msg.parts:
|
||||||
|
if isinstance(part, ToolCallPart):
|
||||||
|
tools_called.append(part.tool_name)
|
||||||
|
elif isinstance(msg, ModelRequest):
|
||||||
|
for part in msg.parts:
|
||||||
|
if isinstance(part, ToolReturnPart):
|
||||||
|
tool_name = part.tool_name
|
||||||
|
content = part.content
|
||||||
|
|
||||||
|
# Categorize as expert result or tool output
|
||||||
|
if tool_name.startswith("delegate_to_"):
|
||||||
|
expert_name = tool_name.replace("delegate_to_", "")
|
||||||
|
expert_results[expert_name] = content
|
||||||
|
else:
|
||||||
|
tool_outputs[tool_name] = content
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"tatlock_orchestration_complete",
|
||||||
|
tools_called=tools_called,
|
||||||
|
expert_count=len(expert_results),
|
||||||
|
tool_output_count=len(tool_outputs),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Add tool-level spans from result messages
|
||||||
|
if orchestrate_span:
|
||||||
|
add_tool_spans_from_messages(result.new_messages(), orchestrate_span)
|
||||||
|
|
||||||
|
# End orchestration span with results
|
||||||
|
end_span(
|
||||||
|
orchestrate_span,
|
||||||
|
metadata_update={
|
||||||
|
"tools_called": tools_called,
|
||||||
|
"expert_count": len(expert_results),
|
||||||
|
"tool_output_count": len(tool_outputs),
|
||||||
|
},
|
||||||
|
details_update={
|
||||||
|
"steward_note_preview": steward_note[:500] if steward_note else None,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"tools_called": tools_called,
|
||||||
|
"expert_results": expert_results,
|
||||||
|
"tool_outputs": tool_outputs,
|
||||||
|
"raw_output": result.output,
|
||||||
|
}
|
||||||
|
|
||||||
|
async def synthesize_from_results(
|
||||||
|
self,
|
||||||
|
user_message: str,
|
||||||
|
orchestration_results: dict[str, Any],
|
||||||
|
message_history: list[dict],
|
||||||
|
) -> str:
|
||||||
|
"""
|
||||||
|
Phase 2: Synthesize butler-toned response from gathered results.
|
||||||
|
|
||||||
|
This is the synthesis phase where Tatlock takes the coordination
|
||||||
|
results and produces a properly butler-toned response.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_message: The user's original message
|
||||||
|
orchestration_results: Results from orchestrate_tool_calls()
|
||||||
|
message_history: Conversation history
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: Butler-toned response synthesized from all results
|
||||||
|
"""
|
||||||
|
from pydantic_ai.messages import (
|
||||||
|
ModelMessage,
|
||||||
|
ModelRequest,
|
||||||
|
ModelResponse,
|
||||||
|
TextPart,
|
||||||
|
UserPromptPart,
|
||||||
|
)
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import get_model
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"tatlock_synthesize_from_results",
|
||||||
|
user_message_preview=user_message[:100],
|
||||||
|
expert_count=len(orchestration_results.get("expert_results", {})),
|
||||||
|
tool_count=len(orchestration_results.get("tool_outputs", {})),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Start tracing span for synthesis phase
|
||||||
|
synthesize_span = start_span(
|
||||||
|
"tatlock_synthesize",
|
||||||
|
SpanType.TATLOCK,
|
||||||
|
metadata={
|
||||||
|
"expert_count": len(orchestration_results.get("expert_results", {})),
|
||||||
|
"tool_output_count": len(orchestration_results.get("tool_outputs", {})),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build synthesis prompt with all available information
|
||||||
|
synthesis_parts = []
|
||||||
|
synthesis_parts.append(f"The user asked: {user_message}")
|
||||||
|
synthesis_parts.append("")
|
||||||
|
|
||||||
|
# Add expert findings if any
|
||||||
|
if orchestration_results.get("expert_results"):
|
||||||
|
synthesis_parts.append("Expert findings:")
|
||||||
|
for expert, result in orchestration_results["expert_results"].items():
|
||||||
|
synthesis_parts.append(f"- {expert.title()}: {result}")
|
||||||
|
synthesis_parts.append("")
|
||||||
|
|
||||||
|
# Add tool outputs if any
|
||||||
|
if orchestration_results.get("tool_outputs"):
|
||||||
|
synthesis_parts.append("Tool results:")
|
||||||
|
for tool, result in orchestration_results["tool_outputs"].items():
|
||||||
|
synthesis_parts.append(f"- {tool}: {result}")
|
||||||
|
synthesis_parts.append("")
|
||||||
|
|
||||||
|
synthesis_parts.append(
|
||||||
|
"Synthesize a response for the user. Be direct and confident. "
|
||||||
|
"Lead with the answer - no apologies, no caveats, no 'mix-ups'. "
|
||||||
|
"Address them as 'sir', be concise, add dry wit if appropriate."
|
||||||
|
)
|
||||||
|
|
||||||
|
synthesis_prompt = "\n".join(synthesis_parts)
|
||||||
|
|
||||||
|
# Create synthesis agent (no tools needed)
|
||||||
|
model = get_model()
|
||||||
|
|
||||||
|
# Synthesis agent uses butler prompt but no tools
|
||||||
|
synthesis_agent = Agent(
|
||||||
|
model,
|
||||||
|
system_prompt=TATLOCK_SYSTEM_PROMPT,
|
||||||
|
# No tools for synthesis phase
|
||||||
|
)
|
||||||
|
|
||||||
|
# Convert message history to PydanticAI format
|
||||||
|
pydantic_history: list[ModelMessage] = []
|
||||||
|
for msg in message_history:
|
||||||
|
role = msg.get("role")
|
||||||
|
content = msg.get("content", "")
|
||||||
|
|
||||||
|
if not content or not content.strip():
|
||||||
|
continue
|
||||||
|
|
||||||
|
if role == "user":
|
||||||
|
pydantic_history.append(ModelRequest(parts=[UserPromptPart(content=content)]))
|
||||||
|
elif role == "assistant":
|
||||||
|
pydantic_history.append(ModelResponse(parts=[TextPart(content=content)]))
|
||||||
|
|
||||||
|
# Run synthesis
|
||||||
|
result = await synthesis_agent.run(
|
||||||
|
synthesis_prompt,
|
||||||
|
message_history=pydantic_history if pydantic_history else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"tatlock_synthesis_complete",
|
||||||
|
response_preview=result.output[:100],
|
||||||
|
)
|
||||||
|
|
||||||
|
# End synthesis span with result
|
||||||
|
end_span(
|
||||||
|
synthesize_span,
|
||||||
|
metadata_update={
|
||||||
|
"response_length": len(result.output),
|
||||||
|
},
|
||||||
|
details_update={
|
||||||
|
"synthesis_prompt": synthesis_prompt[:1000],
|
||||||
|
"response_preview": result.output[:500],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
return result.output
|
||||||
|
|
||||||
async def get_capabilities(self) -> dict:
|
async def get_capabilities(self) -> dict:
|
||||||
"""Return current capabilities."""
|
"""Return current capabilities."""
|
||||||
return {
|
return {
|
||||||
"streaming": True, # Streaming implemented
|
"streaming": True, # Streaming implemented
|
||||||
"reasoning": True, # Basic reasoning summaries
|
"reasoning": True, # Basic reasoning summaries
|
||||||
"tools": True, # Permanent tools: calculator, date/time, search
|
"tools": True, # Permanent tools: calculator, date/time, search
|
||||||
"vision": False, # Future
|
"vision": False, # Future
|
||||||
"audio": False, # Future
|
"audio": False, # Future
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,18 +1,19 @@
|
|||||||
"""
|
"""
|
||||||
Tatlock's core tools package.
|
Tatlock's core tools package.
|
||||||
|
|
||||||
Provides calculator, date/time, and web search capabilities.
|
Provides calculator and date/time capabilities.
|
||||||
|
Web search has been moved to The Librarian agent.
|
||||||
Organized as a household member with toolset and capability registration.
|
Organized as a household member with toolset and capability registration.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from .capability import TATLOCK_CORE_CAPABILITY, get_capability
|
from .capability import TATLOCK_CORE_CAPABILITY, get_capability
|
||||||
from .toolset import get_core_tools, tatlock_core_tools
|
|
||||||
from .tools import (
|
from .tools import (
|
||||||
calculate,
|
calculate,
|
||||||
calculate_time_offset,
|
calculate_time_offset,
|
||||||
get_current_datetime,
|
get_current_datetime,
|
||||||
search_web,
|
|
||||||
time_difference,
|
time_difference,
|
||||||
)
|
)
|
||||||
|
from .toolset import get_core_tools, tatlock_core_tools
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
# Tools
|
# Tools
|
||||||
@@ -20,7 +21,6 @@ __all__ = [
|
|||||||
"get_current_datetime",
|
"get_current_datetime",
|
||||||
"calculate_time_offset",
|
"calculate_time_offset",
|
||||||
"time_difference",
|
"time_difference",
|
||||||
"search_web",
|
|
||||||
# Toolset
|
# Toolset
|
||||||
"tatlock_core_tools",
|
"tatlock_core_tools",
|
||||||
"get_core_tools",
|
"get_core_tools",
|
||||||
|
|||||||
@@ -4,17 +4,17 @@ Household capability definition for Tatlock's core tools.
|
|||||||
Provides the executive summary that the Steward and Butler see
|
Provides the executive summary that the Steward and Butler see
|
||||||
for coordinating household capabilities.
|
for coordinating household capabilities.
|
||||||
"""
|
"""
|
||||||
from src.core.household_registry import HouseholdCapability
|
|
||||||
|
|
||||||
|
from src.core.household_registry import HouseholdCapability
|
||||||
|
|
||||||
TATLOCK_CORE_CAPABILITY = HouseholdCapability(
|
TATLOCK_CORE_CAPABILITY = HouseholdCapability(
|
||||||
name="tatlock_core",
|
name="tatlock_core",
|
||||||
role="Butler's Core Tools",
|
role="Butler's Core Tools",
|
||||||
category="core",
|
category="core",
|
||||||
description="Essential tools for computation, date/time operations, and web searches",
|
description="Essential tools for computation and date/time operations",
|
||||||
domains=["computation", "datetime", "information", "research"],
|
domains=["computation", "datetime", "math", "calculator"],
|
||||||
cost="low",
|
cost="low",
|
||||||
requires_network=True, # For web search
|
requires_network=False, # Web search moved to Librarian
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -6,13 +6,11 @@ These tools are always available to the butler agent:
|
|||||||
- Date/Time toolkit: For current time and time calculations
|
- Date/Time toolkit: For current time and time calculations
|
||||||
- SearXNG search: For searching the web for current information
|
- SearXNG search: For searching the web for current information
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import math
|
import math
|
||||||
import re
|
import re
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
|
|
||||||
import httpx
|
|
||||||
|
|
||||||
from src.core.config import config
|
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
@@ -22,6 +20,7 @@ logger = get_logger(__name__)
|
|||||||
# Calculator Tool
|
# Calculator Tool
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
def calculate(expression: str) -> str:
|
def calculate(expression: str) -> str:
|
||||||
"""
|
"""
|
||||||
Safely evaluate mathematical expressions.
|
Safely evaluate mathematical expressions.
|
||||||
@@ -50,33 +49,29 @@ def calculate(expression: str) -> str:
|
|||||||
# Create safe namespace with math functions
|
# Create safe namespace with math functions
|
||||||
safe_dict = {
|
safe_dict = {
|
||||||
# Basic math functions
|
# Basic math functions
|
||||||
'sqrt': math.sqrt,
|
"sqrt": math.sqrt,
|
||||||
'pow': math.pow,
|
"pow": math.pow,
|
||||||
'abs': abs,
|
"abs": abs,
|
||||||
'round': round,
|
"round": round,
|
||||||
|
|
||||||
# Trigonometric
|
# Trigonometric
|
||||||
'sin': math.sin,
|
"sin": math.sin,
|
||||||
'cos': math.cos,
|
"cos": math.cos,
|
||||||
'tan': math.tan,
|
"tan": math.tan,
|
||||||
'asin': math.asin,
|
"asin": math.asin,
|
||||||
'acos': math.acos,
|
"acos": math.acos,
|
||||||
'atan': math.atan,
|
"atan": math.atan,
|
||||||
|
|
||||||
# Logarithmic
|
# Logarithmic
|
||||||
'log': math.log,
|
"log": math.log,
|
||||||
'log10': math.log10,
|
"log10": math.log10,
|
||||||
'log2': math.log2,
|
"log2": math.log2,
|
||||||
'exp': math.exp,
|
"exp": math.exp,
|
||||||
|
|
||||||
# Other
|
# Other
|
||||||
'ceil': math.ceil,
|
"ceil": math.ceil,
|
||||||
'floor': math.floor,
|
"floor": math.floor,
|
||||||
'factorial': math.factorial,
|
"factorial": math.factorial,
|
||||||
|
|
||||||
# Constants
|
# Constants
|
||||||
'pi': math.pi,
|
"pi": math.pi,
|
||||||
'e': math.e,
|
"e": math.e,
|
||||||
}
|
}
|
||||||
|
|
||||||
# Evaluate the expression safely
|
# Evaluate the expression safely
|
||||||
@@ -101,6 +96,7 @@ def calculate(expression: str) -> str:
|
|||||||
# Date/Time Toolkit
|
# Date/Time Toolkit
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
def get_current_datetime(format_str: str = "full") -> str:
|
def get_current_datetime(format_str: str = "full") -> str:
|
||||||
"""
|
"""
|
||||||
Get the current date and time.
|
Get the current date and time.
|
||||||
@@ -161,7 +157,7 @@ def calculate_time_offset(offset_description: str) -> str:
|
|||||||
|
|
||||||
# Parse the offset description
|
# Parse the offset description
|
||||||
# Pattern: "N unit(s) ago/from now"
|
# Pattern: "N unit(s) ago/from now"
|
||||||
pattern = r'(\d+)\s+(second|minute|hour|day|week|month|year)s?\s+(ago|from\s+now)'
|
pattern = r"(\d+)\s+(second|minute|hour|day|week|month|year)s?\s+(ago|from\s+now)"
|
||||||
match = re.match(pattern, offset_description.lower().strip())
|
match = re.match(pattern, offset_description.lower().strip())
|
||||||
|
|
||||||
if not match:
|
if not match:
|
||||||
@@ -256,96 +252,5 @@ def time_difference(date1_str: str, date2_str: str = "now") -> str:
|
|||||||
return f"Error calculating time difference: {str(e)}"
|
return f"Error calculating time difference: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
# ============================================================================
|
# NOTE: Web search has been moved to The Librarian agent.
|
||||||
# SearXNG Search Tool
|
# Use delegate_to_librarian(task="search web for ...") for web search.
|
||||||
# ============================================================================
|
|
||||||
|
|
||||||
async def search_web(query: str, num_results: int = 5) -> str:
|
|
||||||
"""
|
|
||||||
Search the web using SearXNG.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
query: Search query string
|
|
||||||
num_results: Number of results to return (default: 5, max: 10)
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Formatted search results as a string with titles, URLs, and snippets
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
search_web("Python async programming") -> "1. Title: ...\n URL: ...\n ..."
|
|
||||||
"""
|
|
||||||
try:
|
|
||||||
# Limit results
|
|
||||||
num_results = min(num_results, 10)
|
|
||||||
|
|
||||||
# Get SearXNG host with fallback logic
|
|
||||||
searxng_host = str(config.SEARXNG_HOST)
|
|
||||||
|
|
||||||
# Try production host first, fall back to localhost in development
|
|
||||||
hosts_to_try = [searxng_host]
|
|
||||||
if config.ENVIRONMENT.value == "development" and "localhost" not in searxng_host:
|
|
||||||
# Add localhost fallback for development
|
|
||||||
hosts_to_try.append("http://localhost:8087")
|
|
||||||
|
|
||||||
last_error = None
|
|
||||||
|
|
||||||
for host in hosts_to_try:
|
|
||||||
try:
|
|
||||||
logger.debug("searxng_search_attempt", host=host, query=query)
|
|
||||||
|
|
||||||
async with httpx.AsyncClient(timeout=config.SEARXNG_TIMEOUT) as client:
|
|
||||||
response = await client.get(
|
|
||||||
f"{host}/search",
|
|
||||||
params={
|
|
||||||
"q": query,
|
|
||||||
"format": "json",
|
|
||||||
"pageno": 1,
|
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
if response.status_code == 200:
|
|
||||||
data = response.json()
|
|
||||||
results = data.get("results", [])
|
|
||||||
|
|
||||||
if not results:
|
|
||||||
return f"No results found for '{query}'"
|
|
||||||
|
|
||||||
# Format results
|
|
||||||
formatted_results = []
|
|
||||||
for i, result in enumerate(results[:num_results], 1):
|
|
||||||
title = result.get("title", "No title")
|
|
||||||
url = result.get("url", "")
|
|
||||||
content = result.get("content", "No description available")
|
|
||||||
|
|
||||||
formatted_results.append(
|
|
||||||
f"{i}. {title}\n"
|
|
||||||
f" URL: {url}\n"
|
|
||||||
f" {content}\n"
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
"searxng_search_success",
|
|
||||||
host=host,
|
|
||||||
query=query,
|
|
||||||
result_count=len(results),
|
|
||||||
)
|
|
||||||
return "\n".join(formatted_results)
|
|
||||||
else:
|
|
||||||
last_error = f"SearXNG returned status {response.status_code}"
|
|
||||||
|
|
||||||
except httpx.ConnectError:
|
|
||||||
last_error = f"Cannot connect to SearXNG at {host}"
|
|
||||||
logger.warning("searxng_connection_failed", host=host)
|
|
||||||
continue
|
|
||||||
except Exception as e:
|
|
||||||
last_error = str(e)
|
|
||||||
logger.warning("searxng_error", host=host, error=str(e))
|
|
||||||
continue
|
|
||||||
|
|
||||||
# All hosts failed
|
|
||||||
logger.error("searxng_all_hosts_failed", error=last_error)
|
|
||||||
return f"Error searching: {last_error}. Please check that SearXNG is running."
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error("searxng_unexpected_error", error=str(e), exc_info=True)
|
|
||||||
return f"Error searching: {str(e)}"
|
|
||||||
|
|||||||
@@ -4,11 +4,11 @@ PydanticAI toolset for Tatlock's core tools.
|
|||||||
Converts the core tool functions into PydanticAI tool definitions
|
Converts the core tool functions into PydanticAI tool definitions
|
||||||
that can be registered with agents and the household registry.
|
that can be registered with agents and the household registry.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from pydantic_ai.tools import Tool
|
from pydantic_ai.tools import Tool
|
||||||
|
|
||||||
from . import tools
|
from . import tools
|
||||||
|
|
||||||
|
|
||||||
# Create tool definitions for PydanticAI
|
# Create tool definitions for PydanticAI
|
||||||
calculator_tool = Tool(
|
calculator_tool = Tool(
|
||||||
function=tools.calculate,
|
function=tools.calculate,
|
||||||
@@ -55,17 +55,8 @@ time_difference_tool = Tool(
|
|||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
web_search_tool = Tool(
|
# NOTE: Web search has been moved to The Librarian agent.
|
||||||
function=tools.search_web,
|
# Use delegate_to_librarian(task="search web for ...") for web search.
|
||||||
name="search_web",
|
|
||||||
description=(
|
|
||||||
"Search the web using SearXNG for current information. "
|
|
||||||
"Use this to find recent events, current data, or verify facts. "
|
|
||||||
"Returns formatted results with titles, URLs, and snippets. "
|
|
||||||
"Useful for information that may have changed since training data."
|
|
||||||
),
|
|
||||||
takes_ctx=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
# Combined toolset of all core tools
|
# Combined toolset of all core tools
|
||||||
@@ -74,7 +65,6 @@ tatlock_core_tools = [
|
|||||||
current_datetime_tool,
|
current_datetime_tool,
|
||||||
time_offset_tool,
|
time_offset_tool,
|
||||||
time_difference_tool,
|
time_difference_tool,
|
||||||
web_search_tool,
|
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+25
-122
@@ -4,26 +4,20 @@ Tatlock's permanent tools.
|
|||||||
These tools are always available to the butler agent:
|
These tools are always available to the butler agent:
|
||||||
- Calculator: For all mathematical operations
|
- Calculator: For all mathematical operations
|
||||||
- Date/Time toolkit: For current time and time calculations
|
- Date/Time toolkit: For current time and time calculations
|
||||||
- SearXNG search: For searching the web for current information
|
|
||||||
|
Note: Web search has been moved to The Librarian agent.
|
||||||
|
See src/agents/librarian/tools.py for search_web functionality.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
|
||||||
import math
|
import math
|
||||||
import re
|
import re
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
import httpx
|
|
||||||
|
|
||||||
from src.core.config import config
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
# Calculator Tool
|
# Calculator Tool
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
def calculate(expression: str) -> str:
|
def calculate(expression: str) -> str:
|
||||||
"""
|
"""
|
||||||
Safely evaluate mathematical expressions.
|
Safely evaluate mathematical expressions.
|
||||||
@@ -52,33 +46,29 @@ def calculate(expression: str) -> str:
|
|||||||
# Create safe namespace with math functions
|
# Create safe namespace with math functions
|
||||||
safe_dict = {
|
safe_dict = {
|
||||||
# Basic math functions
|
# Basic math functions
|
||||||
'sqrt': math.sqrt,
|
"sqrt": math.sqrt,
|
||||||
'pow': math.pow,
|
"pow": math.pow,
|
||||||
'abs': abs,
|
"abs": abs,
|
||||||
'round': round,
|
"round": round,
|
||||||
|
|
||||||
# Trigonometric
|
# Trigonometric
|
||||||
'sin': math.sin,
|
"sin": math.sin,
|
||||||
'cos': math.cos,
|
"cos": math.cos,
|
||||||
'tan': math.tan,
|
"tan": math.tan,
|
||||||
'asin': math.asin,
|
"asin": math.asin,
|
||||||
'acos': math.acos,
|
"acos": math.acos,
|
||||||
'atan': math.atan,
|
"atan": math.atan,
|
||||||
|
|
||||||
# Logarithmic
|
# Logarithmic
|
||||||
'log': math.log,
|
"log": math.log,
|
||||||
'log10': math.log10,
|
"log10": math.log10,
|
||||||
'log2': math.log2,
|
"log2": math.log2,
|
||||||
'exp': math.exp,
|
"exp": math.exp,
|
||||||
|
|
||||||
# Other
|
# Other
|
||||||
'ceil': math.ceil,
|
"ceil": math.ceil,
|
||||||
'floor': math.floor,
|
"floor": math.floor,
|
||||||
'factorial': math.factorial,
|
"factorial": math.factorial,
|
||||||
|
|
||||||
# Constants
|
# Constants
|
||||||
'pi': math.pi,
|
"pi": math.pi,
|
||||||
'e': math.e,
|
"e": math.e,
|
||||||
}
|
}
|
||||||
|
|
||||||
# Evaluate the expression safely
|
# Evaluate the expression safely
|
||||||
@@ -103,6 +93,7 @@ def calculate(expression: str) -> str:
|
|||||||
# Date/Time Toolkit
|
# Date/Time Toolkit
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
|
|
||||||
|
|
||||||
def get_current_datetime(format_str: str = "full") -> str:
|
def get_current_datetime(format_str: str = "full") -> str:
|
||||||
"""
|
"""
|
||||||
Get the current date and time.
|
Get the current date and time.
|
||||||
@@ -163,7 +154,7 @@ def calculate_time_offset(offset_description: str) -> str:
|
|||||||
|
|
||||||
# Parse the offset description
|
# Parse the offset description
|
||||||
# Pattern: "N unit(s) ago/from now"
|
# Pattern: "N unit(s) ago/from now"
|
||||||
pattern = r'(\d+)\s+(second|minute|hour|day|week|month|year)s?\s+(ago|from\s+now)'
|
pattern = r"(\d+)\s+(second|minute|hour|day|week|month|year)s?\s+(ago|from\s+now)"
|
||||||
match = re.match(pattern, offset_description.lower().strip())
|
match = re.match(pattern, offset_description.lower().strip())
|
||||||
|
|
||||||
if not match:
|
if not match:
|
||||||
@@ -256,91 +247,3 @@ def time_difference(date1_str: str, date2_str: str = "now") -> str:
|
|||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
return f"Error calculating time difference: {str(e)}"
|
return f"Error calculating time difference: {str(e)}"
|
||||||
|
|
||||||
|
|
||||||
# ============================================================================
|
|
||||||
# SearXNG Search Tool
|
|
||||||
# ============================================================================
|
|
||||||
|
|
||||||
async def search_web(query: str, num_results: int = 5) -> str:
|
|
||||||
"""
|
|
||||||
Search the web using SearXNG.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
query: Search query string
|
|
||||||
num_results: Number of results to return (default: 5, max: 10)
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Formatted search results as a string with titles, URLs, and snippets
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
search_web("Python async programming") -> "1. Title: ...\n URL: ...\n ..."
|
|
||||||
"""
|
|
||||||
try:
|
|
||||||
# Limit results
|
|
||||||
num_results = min(num_results, 10)
|
|
||||||
|
|
||||||
# Get SearXNG host with fallback logic
|
|
||||||
searxng_host = str(config.SEARXNG_HOST)
|
|
||||||
|
|
||||||
# Try production host first, fall back to localhost in development
|
|
||||||
hosts_to_try = [searxng_host]
|
|
||||||
if config.ENVIRONMENT.value == "development" and "localhost" not in searxng_host:
|
|
||||||
# Add localhost fallback for development
|
|
||||||
hosts_to_try.append("http://localhost:8087")
|
|
||||||
|
|
||||||
last_error = None
|
|
||||||
|
|
||||||
for host in hosts_to_try:
|
|
||||||
try:
|
|
||||||
logger.info(f"Attempting SearXNG search at {host}")
|
|
||||||
|
|
||||||
async with httpx.AsyncClient(timeout=config.SEARXNG_TIMEOUT) as client:
|
|
||||||
response = await client.get(
|
|
||||||
f"{host}/search",
|
|
||||||
params={
|
|
||||||
"q": query,
|
|
||||||
"format": "json",
|
|
||||||
"pageno": 1,
|
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
if response.status_code == 200:
|
|
||||||
data = response.json()
|
|
||||||
results = data.get("results", [])
|
|
||||||
|
|
||||||
if not results:
|
|
||||||
return f"No results found for '{query}'"
|
|
||||||
|
|
||||||
# Format results
|
|
||||||
formatted_results = []
|
|
||||||
for i, result in enumerate(results[:num_results], 1):
|
|
||||||
title = result.get("title", "No title")
|
|
||||||
url = result.get("url", "")
|
|
||||||
content = result.get("content", "No description available")
|
|
||||||
|
|
||||||
formatted_results.append(
|
|
||||||
f"{i}. {title}\n"
|
|
||||||
f" URL: {url}\n"
|
|
||||||
f" {content}\n"
|
|
||||||
)
|
|
||||||
|
|
||||||
return "\n".join(formatted_results)
|
|
||||||
else:
|
|
||||||
last_error = f"SearXNG returned status {response.status_code}"
|
|
||||||
|
|
||||||
except httpx.ConnectError:
|
|
||||||
last_error = f"Cannot connect to SearXNG at {host}"
|
|
||||||
logger.warning(f"SearXNG connection failed at {host}, trying next host if available")
|
|
||||||
continue
|
|
||||||
except Exception as e:
|
|
||||||
last_error = str(e)
|
|
||||||
logger.warning(f"SearXNG error at {host}: {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
# All hosts failed
|
|
||||||
return f"Error searching: {last_error}. Please check that SearXNG is running."
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Unexpected error in search_web: {e}", exc_info=True)
|
|
||||||
return f"Error searching: {str(e)}"
|
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
"""
|
||||||
|
Anthropic/Claude integration module.
|
||||||
|
|
||||||
|
Provides model selection with Ollama as primary backend and Claude
|
||||||
|
as the cloud fallback.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from src.anthropic.model_selector import (
|
||||||
|
check_claude_health,
|
||||||
|
check_ollama_health,
|
||||||
|
get_model,
|
||||||
|
get_tool_choice_settings,
|
||||||
|
is_claude_available,
|
||||||
|
is_ollama_available,
|
||||||
|
resolve_backend,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"check_claude_health",
|
||||||
|
"check_ollama_health",
|
||||||
|
"get_model",
|
||||||
|
"get_tool_choice_settings",
|
||||||
|
"is_claude_available",
|
||||||
|
"is_ollama_available",
|
||||||
|
"resolve_backend",
|
||||||
|
]
|
||||||
@@ -0,0 +1,293 @@
|
|||||||
|
"""
|
||||||
|
Model selector for Ollama/Claude backend switching.
|
||||||
|
|
||||||
|
Provides automatic model selection with Ollama as the primary local backend
|
||||||
|
and Claude as the cloud fallback. Claude is used when PREFER_CLOUD_BACKEND
|
||||||
|
is enabled, or automatically when Ollama is unavailable at startup.
|
||||||
|
|
||||||
|
The Anthropic SDK is imported lazily so a missing or broken `anthropic`
|
||||||
|
package degrades to Ollama-only operation instead of crashing the app.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from src.core.config import config
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from pydantic_ai.models.anthropic import AnthropicModel
|
||||||
|
from pydantic_ai.models.openai import OpenAIChatModel
|
||||||
|
from pydantic_ai.settings import ModelSettings
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
# Cached health check results (set once at startup)
|
||||||
|
_claude_available: bool | None = None
|
||||||
|
_ollama_available: bool | None = None
|
||||||
|
|
||||||
|
|
||||||
|
async def check_ollama_health() -> bool:
|
||||||
|
"""
|
||||||
|
Check if the Ollama server is reachable and has the configured model.
|
||||||
|
|
||||||
|
This should be called once at application startup.
|
||||||
|
The result is cached in `_ollama_available`.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if Ollama is reachable and OLLAMA_DEFAULT_MODEL is pulled.
|
||||||
|
"""
|
||||||
|
global _ollama_available
|
||||||
|
|
||||||
|
host = str(config.OLLAMA_HOST).rstrip("/")
|
||||||
|
model = config.OLLAMA_DEFAULT_MODEL
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||||
|
response = await client.get(f"{host}/api/tags")
|
||||||
|
response.raise_for_status()
|
||||||
|
names = [m.get("name", "") for m in response.json().get("models", [])]
|
||||||
|
|
||||||
|
if model in names or f"{model}:latest" in names:
|
||||||
|
_ollama_available = True
|
||||||
|
logger.info(
|
||||||
|
"ollama_health_check_passed",
|
||||||
|
host=host,
|
||||||
|
model=model,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
_ollama_available = False
|
||||||
|
logger.warning(
|
||||||
|
"ollama_health_check_failed",
|
||||||
|
reason="model_not_pulled",
|
||||||
|
host=host,
|
||||||
|
model=model,
|
||||||
|
hint=f"run `ollama pull {model}`",
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
_ollama_available = False
|
||||||
|
logger.warning(
|
||||||
|
"ollama_health_check_failed",
|
||||||
|
reason="server_unreachable",
|
||||||
|
host=host,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
async def check_claude_health() -> bool:
|
||||||
|
"""
|
||||||
|
Check if Claude API is reachable and working.
|
||||||
|
|
||||||
|
This should be called once at application startup.
|
||||||
|
The result is cached in `_claude_available`.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if Claude API is accessible, False otherwise.
|
||||||
|
"""
|
||||||
|
global _claude_available
|
||||||
|
|
||||||
|
# No API key configured - Claude not available
|
||||||
|
if not config.ANTHROPIC_API_KEY:
|
||||||
|
logger.info(
|
||||||
|
"claude_health_check_skipped",
|
||||||
|
reason="no_api_key",
|
||||||
|
)
|
||||||
|
_claude_available = False
|
||||||
|
return False
|
||||||
|
|
||||||
|
try:
|
||||||
|
from anthropic import AsyncAnthropic
|
||||||
|
|
||||||
|
client = AsyncAnthropic(api_key=config.ANTHROPIC_API_KEY)
|
||||||
|
|
||||||
|
# Minimal API call to verify connectivity
|
||||||
|
# Using a tiny max_tokens to minimize cost
|
||||||
|
await client.messages.create(
|
||||||
|
model=config.ANTHROPIC_MODEL,
|
||||||
|
max_tokens=1,
|
||||||
|
messages=[{"role": "user", "content": "hi"}],
|
||||||
|
)
|
||||||
|
|
||||||
|
_claude_available = True
|
||||||
|
logger.info(
|
||||||
|
"claude_health_check_passed",
|
||||||
|
model=config.ANTHROPIC_MODEL,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
_claude_available = False
|
||||||
|
logger.warning(
|
||||||
|
"claude_health_check_failed",
|
||||||
|
error=str(e),
|
||||||
|
model=config.ANTHROPIC_MODEL,
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def is_claude_available() -> bool:
|
||||||
|
"""
|
||||||
|
Check if Claude is available (from cached health check result).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if Claude API was reachable at startup, False otherwise.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
Returns False if health check hasn't been run yet.
|
||||||
|
Call `check_claude_health()` at startup first.
|
||||||
|
"""
|
||||||
|
return _claude_available is True
|
||||||
|
|
||||||
|
|
||||||
|
def is_ollama_available() -> bool:
|
||||||
|
"""
|
||||||
|
Check if Ollama is available (from cached health check result).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
False only if the startup health check confirmed Ollama is down.
|
||||||
|
Unknown (check not run yet) counts as available so that contexts
|
||||||
|
without lifespan events keep the local-first behavior.
|
||||||
|
"""
|
||||||
|
return _ollama_available is not False
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_backend(prefer_cloud: bool | None = None) -> str:
|
||||||
|
"""
|
||||||
|
Resolve which backend should serve requests.
|
||||||
|
|
||||||
|
Ollama is the primary backend. Claude is used when explicitly
|
||||||
|
preferred via PREFER_CLOUD_BACKEND, or as automatic fallback
|
||||||
|
when the startup health check found Ollama down.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
prefer_cloud: Override config.PREFER_CLOUD_BACKEND for this call.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
"claude" or "ollama".
|
||||||
|
"""
|
||||||
|
use_cloud = prefer_cloud if prefer_cloud is not None else config.PREFER_CLOUD_BACKEND
|
||||||
|
|
||||||
|
if use_cloud and is_claude_available():
|
||||||
|
return "claude"
|
||||||
|
|
||||||
|
if not is_ollama_available() and is_claude_available():
|
||||||
|
logger.warning(
|
||||||
|
"backend_fallback_to_claude",
|
||||||
|
reason="ollama_unavailable",
|
||||||
|
)
|
||||||
|
return "claude"
|
||||||
|
|
||||||
|
return "ollama"
|
||||||
|
|
||||||
|
|
||||||
|
def get_model(prefer_cloud: bool | None = None) -> AnthropicModel | OpenAIChatModel:
|
||||||
|
"""
|
||||||
|
Get the best available model.
|
||||||
|
|
||||||
|
Returns Ollama unless Claude is preferred (or Ollama is down).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
prefer_cloud: Override config.PREFER_CLOUD_BACKEND for this call.
|
||||||
|
If None, uses the config value.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PydanticAI model instance (OpenAIChatModel or AnthropicModel).
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> model = get_model()
|
||||||
|
>>> agent = Agent(model, system_prompt="...")
|
||||||
|
"""
|
||||||
|
if resolve_backend(prefer_cloud) == "claude":
|
||||||
|
try:
|
||||||
|
from pydantic_ai.models.anthropic import AnthropicModel
|
||||||
|
from pydantic_ai.providers.anthropic import AnthropicProvider
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"model_selected",
|
||||||
|
backend="claude",
|
||||||
|
model=config.ANTHROPIC_MODEL,
|
||||||
|
)
|
||||||
|
return AnthropicModel(
|
||||||
|
model_name=config.ANTHROPIC_MODEL,
|
||||||
|
provider=AnthropicProvider(api_key=config.ANTHROPIC_API_KEY),
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
logger.error(
|
||||||
|
"claude_backend_import_failed",
|
||||||
|
error=str(e),
|
||||||
|
hint="anthropic package missing or incompatible; using Ollama",
|
||||||
|
)
|
||||||
|
|
||||||
|
from pydantic_ai.models.openai import OpenAIChatModel
|
||||||
|
|
||||||
|
from src.ollama.provider import get_ollama_provider
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"model_selected",
|
||||||
|
backend="ollama",
|
||||||
|
model=config.OLLAMA_DEFAULT_MODEL,
|
||||||
|
)
|
||||||
|
return OpenAIChatModel(
|
||||||
|
model_name=config.OLLAMA_DEFAULT_MODEL,
|
||||||
|
provider=get_ollama_provider(),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_tool_choice_settings() -> ModelSettings:
|
||||||
|
"""
|
||||||
|
Get model_settings for forcing tool calls on the first request.
|
||||||
|
|
||||||
|
For Claude: PydanticAI handles tool_choice natively, so no extra_body needed.
|
||||||
|
For Ollama: Pass tool_choice="required" via extra_body to force tool calling.
|
||||||
|
"""
|
||||||
|
from pydantic_ai.settings import ModelSettings
|
||||||
|
|
||||||
|
if resolve_backend() == "claude":
|
||||||
|
# PydanticAI's Anthropic model handles tool_choice internally
|
||||||
|
return ModelSettings()
|
||||||
|
else:
|
||||||
|
# Ollama needs explicit tool_choice via extra_body
|
||||||
|
return ModelSettings(extra_body={"tool_choice": "required"})
|
||||||
|
|
||||||
|
|
||||||
|
def get_sampling_settings(temperature: float) -> ModelSettings:
|
||||||
|
"""
|
||||||
|
Get model_settings with a sampling temperature where the backend allows it.
|
||||||
|
|
||||||
|
Ollama accepts a temperature; Claude Sonnet 5+ rejects sampling
|
||||||
|
parameters, so the Claude backend gets empty settings.
|
||||||
|
"""
|
||||||
|
from pydantic_ai.settings import ModelSettings
|
||||||
|
|
||||||
|
if resolve_backend() == "claude":
|
||||||
|
return ModelSettings()
|
||||||
|
return ModelSettings(temperature=temperature)
|
||||||
|
|
||||||
|
|
||||||
|
def get_model_info() -> dict:
|
||||||
|
"""
|
||||||
|
Get information about the current model configuration.
|
||||||
|
|
||||||
|
Useful for health checks and debugging.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with backend, model name, and availability info.
|
||||||
|
"""
|
||||||
|
backend = resolve_backend()
|
||||||
|
|
||||||
|
return {
|
||||||
|
"backend": backend,
|
||||||
|
"model": config.ANTHROPIC_MODEL if backend == "claude" else config.OLLAMA_DEFAULT_MODEL,
|
||||||
|
"claude_available": is_claude_available(),
|
||||||
|
"claude_configured": bool(config.ANTHROPIC_API_KEY),
|
||||||
|
"ollama_available": is_ollama_available(),
|
||||||
|
"ollama_model": config.OLLAMA_DEFAULT_MODEL,
|
||||||
|
"prefer_cloud": config.PREFER_CLOUD_BACKEND,
|
||||||
|
}
|
||||||
+24
-19
@@ -2,12 +2,13 @@
|
|||||||
Chat completion router.
|
Chat completion router.
|
||||||
OpenAI-compatible /v1/chat/completions endpoint.
|
OpenAI-compatible /v1/chat/completions endpoint.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
from typing import AsyncGenerator
|
from collections.abc import AsyncGenerator
|
||||||
|
|
||||||
from fastapi import APIRouter
|
from fastapi import APIRouter
|
||||||
from sse_starlette.sse import EventSourceResponse
|
from starlette.responses import StreamingResponse
|
||||||
|
|
||||||
from src.chat import service
|
from src.chat import service
|
||||||
from src.chat.schemas import (
|
from src.chat.schemas import (
|
||||||
@@ -22,47 +23,51 @@ router = APIRouter(prefix="/chat", tags=["chat"])
|
|||||||
|
|
||||||
async def _stream_response(
|
async def _stream_response(
|
||||||
request: ChatCompletionRequest,
|
request: ChatCompletionRequest,
|
||||||
) -> AsyncGenerator[dict, None]:
|
) -> AsyncGenerator[str, None]:
|
||||||
"""
|
"""
|
||||||
Generate SSE stream for chat completion.
|
Generate SSE stream for chat completion.
|
||||||
|
|
||||||
EventSourceResponse adds "data: " prefix automatically.
|
Yields raw SSE-formatted strings matching OpenAI's format exactly:
|
||||||
We just yield the dict/string content.
|
data: {json}\n\n
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
async for chunk in service.create_chat_completion_stream(request):
|
async for chunk in service.create_chat_completion_stream(request):
|
||||||
# Yield dict - EventSourceResponse will format as SSE
|
yield f"data: {chunk.model_dump_json(exclude_unset=True)}\n\n"
|
||||||
yield {"data": chunk.model_dump_json()}
|
|
||||||
|
|
||||||
# Send [DONE] message
|
yield "data: [DONE]\n\n"
|
||||||
yield {"data": "[DONE]"}
|
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"Error in streaming response: {e}")
|
logger.error(f"Error in streaming response: {e}")
|
||||||
error_data = {"error": {"message": str(e), "type": "internal_error"}}
|
error_data = json.dumps({"error": {"message": str(e), "type": "internal_error"}})
|
||||||
yield {"data": json.dumps(error_data)}
|
yield f"data: {error_data}\n\n"
|
||||||
|
|
||||||
|
|
||||||
@router.post("/completions", response_model=ChatCompletionResponse)
|
@router.post("/completions", response_model=ChatCompletionResponse)
|
||||||
async def create_chat_completion(
|
async def create_chat_completion(
|
||||||
request: ChatCompletionRequest,
|
request: ChatCompletionRequest,
|
||||||
) -> ChatCompletionResponse | EventSourceResponse:
|
) -> ChatCompletionResponse | StreamingResponse:
|
||||||
"""
|
"""
|
||||||
Create chat completion (OpenAI-compatible).
|
Create chat completion (OpenAI-compatible).
|
||||||
|
|
||||||
Supports both regular and streaming responses.
|
Supports both regular and streaming responses.
|
||||||
Currently returns mock lorem ipsum responses.
|
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
request: Chat completion request
|
request: Chat completion request
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Chat completion response or SSE stream
|
Chat completion response or SSE stream
|
||||||
"""
|
"""
|
||||||
logger.info(f"Chat completion request for model: {request.model}")
|
logger.info(f"Chat completion request for model: {request.model}")
|
||||||
|
|
||||||
if request.stream:
|
if request.stream:
|
||||||
logger.info("Streaming response requested")
|
logger.info("Streaming response requested")
|
||||||
return EventSourceResponse(_stream_response(request))
|
return StreamingResponse(
|
||||||
|
_stream_response(request),
|
||||||
|
media_type="text/event-stream",
|
||||||
|
headers={
|
||||||
|
"Cache-Control": "no-store",
|
||||||
|
"X-Accel-Buffering": "no",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
return await service.create_chat_completion(request)
|
return await service.create_chat_completion(request)
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
OpenAI-compatible chat completion schemas.
|
OpenAI-compatible chat completion schemas.
|
||||||
Following OpenAI API specification for compatibility.
|
Following OpenAI API specification for compatibility.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from typing import Literal
|
from typing import Literal
|
||||||
|
|
||||||
from pydantic import Field
|
from pydantic import Field
|
||||||
@@ -11,6 +12,7 @@ from src.core.models import CustomBaseModel
|
|||||||
|
|
||||||
class ChatMessage(CustomBaseModel):
|
class ChatMessage(CustomBaseModel):
|
||||||
"""OpenAI-compatible chat message."""
|
"""OpenAI-compatible chat message."""
|
||||||
|
|
||||||
role: Literal["system", "user", "assistant"]
|
role: Literal["system", "user", "assistant"]
|
||||||
content: str
|
content: str
|
||||||
name: str | None = None
|
name: str | None = None
|
||||||
@@ -18,6 +20,7 @@ class ChatMessage(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionRequest(CustomBaseModel):
|
class ChatCompletionRequest(CustomBaseModel):
|
||||||
"""OpenAI-compatible chat completion request."""
|
"""OpenAI-compatible chat completion request."""
|
||||||
|
|
||||||
model: str = Field(..., description="Model to use for completion")
|
model: str = Field(..., description="Model to use for completion")
|
||||||
messages: list[ChatMessage] = Field(..., description="List of messages")
|
messages: list[ChatMessage] = Field(..., description="List of messages")
|
||||||
temperature: float | None = Field(default=0.7, ge=0.0, le=2.0)
|
temperature: float | None = Field(default=0.7, ge=0.0, le=2.0)
|
||||||
@@ -29,6 +32,7 @@ class ChatCompletionRequest(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionChoice(CustomBaseModel):
|
class ChatCompletionChoice(CustomBaseModel):
|
||||||
"""Choice in chat completion response."""
|
"""Choice in chat completion response."""
|
||||||
|
|
||||||
index: int
|
index: int
|
||||||
message: ChatMessage
|
message: ChatMessage
|
||||||
finish_reason: str | None
|
finish_reason: str | None
|
||||||
@@ -36,6 +40,7 @@ class ChatCompletionChoice(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionUsage(CustomBaseModel):
|
class ChatCompletionUsage(CustomBaseModel):
|
||||||
"""Token usage information."""
|
"""Token usage information."""
|
||||||
|
|
||||||
prompt_tokens: int
|
prompt_tokens: int
|
||||||
completion_tokens: int
|
completion_tokens: int
|
||||||
total_tokens: int
|
total_tokens: int
|
||||||
@@ -43,6 +48,7 @@ class ChatCompletionUsage(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionResponse(CustomBaseModel):
|
class ChatCompletionResponse(CustomBaseModel):
|
||||||
"""OpenAI-compatible chat completion response."""
|
"""OpenAI-compatible chat completion response."""
|
||||||
|
|
||||||
id: str
|
id: str
|
||||||
object: str = "chat.completion"
|
object: str = "chat.completion"
|
||||||
created: int
|
created: int
|
||||||
@@ -53,12 +59,15 @@ class ChatCompletionResponse(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionChunkDelta(CustomBaseModel):
|
class ChatCompletionChunkDelta(CustomBaseModel):
|
||||||
"""Delta in streaming chunk."""
|
"""Delta in streaming chunk."""
|
||||||
|
|
||||||
role: str | None = None
|
role: str | None = None
|
||||||
content: str | None = None
|
content: str | None = None
|
||||||
|
reasoning_content: str | None = None # For thinking/reasoning (DeepSeek R1 format)
|
||||||
|
|
||||||
|
|
||||||
class ChatCompletionChunkChoice(CustomBaseModel):
|
class ChatCompletionChunkChoice(CustomBaseModel):
|
||||||
"""Choice in streaming chunk."""
|
"""Choice in streaming chunk."""
|
||||||
|
|
||||||
index: int
|
index: int
|
||||||
delta: ChatCompletionChunkDelta
|
delta: ChatCompletionChunkDelta
|
||||||
finish_reason: str | None = None
|
finish_reason: str | None = None
|
||||||
@@ -66,6 +75,7 @@ class ChatCompletionChunkChoice(CustomBaseModel):
|
|||||||
|
|
||||||
class ChatCompletionChunk(CustomBaseModel):
|
class ChatCompletionChunk(CustomBaseModel):
|
||||||
"""OpenAI-compatible streaming chunk."""
|
"""OpenAI-compatible streaming chunk."""
|
||||||
|
|
||||||
id: str
|
id: str
|
||||||
object: str = "chat.completion.chunk"
|
object: str = "chat.completion.chunk"
|
||||||
created: int
|
created: int
|
||||||
|
|||||||
+17
-50
@@ -4,17 +4,17 @@ Chat completion service.
|
|||||||
Wrapper around Responses API that converts to Chat Completions format.
|
Wrapper around Responses API that converts to Chat Completions format.
|
||||||
Embeds reasoning in <think> tags for Open WebUI compatibility.
|
Embeds reasoning in <think> tags for Open WebUI compatibility.
|
||||||
"""
|
"""
|
||||||
import asyncio
|
|
||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
from typing import AsyncGenerator
|
from collections.abc import AsyncGenerator
|
||||||
|
|
||||||
from src.chat import constants
|
from src.chat import constants
|
||||||
from src.chat.schemas import (
|
from src.chat.schemas import (
|
||||||
|
ChatCompletionChoice,
|
||||||
ChatCompletionChunk,
|
ChatCompletionChunk,
|
||||||
ChatCompletionChunkChoice,
|
ChatCompletionChunkChoice,
|
||||||
ChatCompletionChunkDelta,
|
ChatCompletionChunkDelta,
|
||||||
ChatCompletionChoice,
|
|
||||||
ChatCompletionRequest,
|
ChatCompletionRequest,
|
||||||
ChatCompletionResponse,
|
ChatCompletionResponse,
|
||||||
ChatCompletionUsage,
|
ChatCompletionUsage,
|
||||||
@@ -43,10 +43,7 @@ async def create_chat_completion(
|
|||||||
created_at = int(time.time())
|
created_at = int(time.time())
|
||||||
|
|
||||||
# Convert Chat request to Responses request
|
# Convert Chat request to Responses request
|
||||||
input_messages = [
|
input_messages = [{"role": msg.role, "content": msg.content} for msg in request.messages]
|
||||||
{"role": msg.role, "content": msg.content}
|
|
||||||
for msg in request.messages
|
|
||||||
]
|
|
||||||
|
|
||||||
response_request = ResponseRequest(
|
response_request = ResponseRequest(
|
||||||
model=request.model,
|
model=request.model,
|
||||||
@@ -54,7 +51,9 @@ async def create_chat_completion(
|
|||||||
reasoning={"effort": "medium", "summary": "auto"}, # Enable reasoning
|
reasoning={"effort": "medium", "summary": "auto"}, # Enable reasoning
|
||||||
temperature=request.temperature or 1.0,
|
temperature=request.temperature or 1.0,
|
||||||
max_output_tokens=request.max_tokens,
|
max_output_tokens=request.max_tokens,
|
||||||
stop=request.stop if isinstance(request.stop, list) else ([request.stop] if request.stop else None),
|
stop=request.stop
|
||||||
|
if isinstance(request.stop, list)
|
||||||
|
else ([request.stop] if request.stop else None),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Call Responses API (will use Steward for Tatlock)
|
# Call Responses API (will use Steward for Tatlock)
|
||||||
@@ -118,16 +117,13 @@ async def create_chat_completion_stream(
|
|||||||
Yields:
|
Yields:
|
||||||
Chat completion chunks with reasoning as <think> tags
|
Chat completion chunks with reasoning as <think> tags
|
||||||
"""
|
"""
|
||||||
from src.responses.streaming import StreamingCoordinator, StreamEventType
|
from src.responses.streaming import StreamEventType, StreamingCoordinator
|
||||||
|
|
||||||
completion_id = f"chatcmpl-{uuid.uuid4().hex[:24]}"
|
completion_id = f"chatcmpl-{uuid.uuid4().hex[:24]}"
|
||||||
created_at = int(time.time())
|
created_at = int(time.time())
|
||||||
|
|
||||||
# Convert Chat request to Responses request
|
# Convert Chat request to Responses request
|
||||||
input_messages = [
|
input_messages = [{"role": msg.role, "content": msg.content} for msg in request.messages]
|
||||||
{"role": msg.role, "content": msg.content}
|
|
||||||
for msg in request.messages
|
|
||||||
]
|
|
||||||
|
|
||||||
response_request = ResponseRequest(
|
response_request = ResponseRequest(
|
||||||
model=request.model,
|
model=request.model,
|
||||||
@@ -135,7 +131,9 @@ async def create_chat_completion_stream(
|
|||||||
reasoning={"effort": "medium", "summary": "auto"},
|
reasoning={"effort": "medium", "summary": "auto"},
|
||||||
temperature=request.temperature or 1.0,
|
temperature=request.temperature or 1.0,
|
||||||
max_output_tokens=request.max_tokens,
|
max_output_tokens=request.max_tokens,
|
||||||
stop=request.stop if isinstance(request.stop, list) else ([request.stop] if request.stop else None),
|
stop=request.stop
|
||||||
|
if isinstance(request.stop, list)
|
||||||
|
else ([request.stop] if request.stop else None),
|
||||||
stream=True,
|
stream=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -163,7 +161,6 @@ async def create_chat_completion_stream(
|
|||||||
|
|
||||||
# Stream from Responses API
|
# Stream from Responses API
|
||||||
coordinator = StreamingCoordinator()
|
coordinator = StreamingCoordinator()
|
||||||
in_reasoning = False
|
|
||||||
|
|
||||||
if use_steward:
|
if use_steward:
|
||||||
stream_generator = coordinator.stream_response_with_steward(response_request)
|
stream_generator = coordinator.stream_response_with_steward(response_request)
|
||||||
@@ -172,24 +169,8 @@ async def create_chat_completion_stream(
|
|||||||
|
|
||||||
async for event in stream_generator:
|
async for event in stream_generator:
|
||||||
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
|
if event.event == StreamEventType.REASONING_SUMMARY_DELTA:
|
||||||
# Start <think> block if needed
|
# Stream reasoning via reasoning_content field (DeepSeek R1 format)
|
||||||
if not in_reasoning:
|
# Open WebUI renders this as collapsible thinking block
|
||||||
yield ChatCompletionChunk(
|
|
||||||
id=completion_id,
|
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
|
||||||
created=created_at,
|
|
||||||
model=request.model,
|
|
||||||
choices=[
|
|
||||||
ChatCompletionChunkChoice(
|
|
||||||
index=0,
|
|
||||||
delta=ChatCompletionChunkDelta(content="<think>\n"),
|
|
||||||
finish_reason=None,
|
|
||||||
)
|
|
||||||
],
|
|
||||||
)
|
|
||||||
in_reasoning = True
|
|
||||||
|
|
||||||
# Stream reasoning delta
|
|
||||||
yield ChatCompletionChunk(
|
yield ChatCompletionChunk(
|
||||||
id=completion_id,
|
id=completion_id,
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
||||||
@@ -198,29 +179,15 @@ async def create_chat_completion_stream(
|
|||||||
choices=[
|
choices=[
|
||||||
ChatCompletionChunkChoice(
|
ChatCompletionChunkChoice(
|
||||||
index=0,
|
index=0,
|
||||||
delta=ChatCompletionChunkDelta(content=event.delta),
|
delta=ChatCompletionChunkDelta(reasoning_content=event.delta),
|
||||||
finish_reason=None,
|
finish_reason=None,
|
||||||
)
|
)
|
||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
elif event.event == StreamEventType.REASONING_SUMMARY_DONE:
|
elif event.event == StreamEventType.REASONING_SUMMARY_DONE:
|
||||||
# Close <think> block
|
# Signal end of reasoning block (no content needed)
|
||||||
if in_reasoning:
|
pass # nothing downstream reads this; the event just ends the block
|
||||||
yield ChatCompletionChunk(
|
|
||||||
id=completion_id,
|
|
||||||
object=constants.CHAT_COMPLETION_CHUNK_OBJECT,
|
|
||||||
created=created_at,
|
|
||||||
model=request.model,
|
|
||||||
choices=[
|
|
||||||
ChatCompletionChunkChoice(
|
|
||||||
index=0,
|
|
||||||
delta=ChatCompletionChunkDelta(content="</think>\n\n"),
|
|
||||||
finish_reason=None,
|
|
||||||
)
|
|
||||||
],
|
|
||||||
)
|
|
||||||
in_reasoning = False
|
|
||||||
|
|
||||||
elif event.event == StreamEventType.OUTPUT_TEXT_DELTA:
|
elif event.event == StreamEventType.OUTPUT_TEXT_DELTA:
|
||||||
# Stream message content
|
# Stream message content
|
||||||
|
|||||||
@@ -1,337 +0,0 @@
|
|||||||
"""
|
|
||||||
Performance benchmark storage using Redis.
|
|
||||||
|
|
||||||
Tracks operation timing, tool usage, and recommendation accuracy across sessions.
|
|
||||||
Provides time-series data for performance analysis and optimization.
|
|
||||||
"""
|
|
||||||
import json
|
|
||||||
from datetime import datetime, timezone
|
|
||||||
from typing import Any, Literal, Optional
|
|
||||||
|
|
||||||
import redis.asyncio as redis
|
|
||||||
from pydantic import BaseModel, Field
|
|
||||||
|
|
||||||
from .config import config
|
|
||||||
from .logging_config import get_logger
|
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
class PerformanceBenchmark(BaseModel):
|
|
||||||
"""
|
|
||||||
Performance benchmark record.
|
|
||||||
|
|
||||||
Stores timing and metadata for operations like Steward analysis,
|
|
||||||
tool calls, and agent execution.
|
|
||||||
"""
|
|
||||||
timestamp: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
||||||
operation: str # "steward_analysis", "tool_call", "tatlock_execution"
|
|
||||||
duration_seconds: float
|
|
||||||
success: bool
|
|
||||||
|
|
||||||
# Steward-specific fields
|
|
||||||
recommendation_count: Optional[int] = None
|
|
||||||
confidence: Optional[float] = None
|
|
||||||
|
|
||||||
# Tool-specific fields
|
|
||||||
tool_name: Optional[str] = None
|
|
||||||
was_recommended: Optional[bool] = None
|
|
||||||
was_actually_used: Optional[bool] = None
|
|
||||||
|
|
||||||
# Context
|
|
||||||
conversation_id: Optional[str] = None
|
|
||||||
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
||||||
|
|
||||||
def to_redis_dict(self) -> dict[str, Any]:
|
|
||||||
"""Convert to dict suitable for Redis storage."""
|
|
||||||
data = self.model_dump()
|
|
||||||
data["timestamp"] = self.timestamp.isoformat()
|
|
||||||
data["metadata"] = json.dumps(self.metadata)
|
|
||||||
return data
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def from_redis_dict(cls, data: dict[str, Any]) -> "PerformanceBenchmark":
|
|
||||||
"""Reconstruct from Redis dict."""
|
|
||||||
data["timestamp"] = datetime.fromisoformat(data["timestamp"])
|
|
||||||
data["metadata"] = json.loads(data.get("metadata", "{}"))
|
|
||||||
return cls(**data)
|
|
||||||
|
|
||||||
|
|
||||||
class BenchmarkStore:
|
|
||||||
"""
|
|
||||||
Redis-backed benchmark storage with automatic expiry.
|
|
||||||
|
|
||||||
Stores performance metrics in time-series format with 30-day retention.
|
|
||||||
Provides querying capabilities for analysis and reporting.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, redis_client: Optional[redis.Redis] = None):
|
|
||||||
"""
|
|
||||||
Initialize benchmark store.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
redis_client: Optional Redis client. If None, creates from config.
|
|
||||||
"""
|
|
||||||
self._client = redis_client
|
|
||||||
self._ttl_days = 30 # 30-day retention
|
|
||||||
|
|
||||||
async def _get_client(self) -> redis.Redis:
|
|
||||||
"""Get or create Redis client."""
|
|
||||||
if self._client is None:
|
|
||||||
self._client = redis.from_url(
|
|
||||||
config.redis_url,
|
|
||||||
encoding="utf-8",
|
|
||||||
decode_responses=True,
|
|
||||||
socket_timeout=config.REDIS_TIMEOUT,
|
|
||||||
socket_connect_timeout=config.REDIS_TIMEOUT,
|
|
||||||
)
|
|
||||||
return self._client
|
|
||||||
|
|
||||||
async def record(self, benchmark: PerformanceBenchmark) -> None:
|
|
||||||
"""
|
|
||||||
Record a performance benchmark.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
benchmark: Performance benchmark to record
|
|
||||||
|
|
||||||
Example:
|
|
||||||
>>> await store.record(PerformanceBenchmark(
|
|
||||||
... operation="steward_analysis",
|
|
||||||
... duration_seconds=1.23,
|
|
||||||
... success=True,
|
|
||||||
... recommendation_count=3,
|
|
||||||
... ))
|
|
||||||
"""
|
|
||||||
if not config.ENABLE_BENCHMARKS:
|
|
||||||
return
|
|
||||||
|
|
||||||
try:
|
|
||||||
client = await self._get_client()
|
|
||||||
|
|
||||||
# Generate key: benchmark:{operation}:{timestamp_ms}
|
|
||||||
timestamp_ms = int(benchmark.timestamp.timestamp() * 1000)
|
|
||||||
key = f"benchmark:{benchmark.operation}:{timestamp_ms}"
|
|
||||||
|
|
||||||
# Store as hash
|
|
||||||
await client.hset(key, mapping=benchmark.to_redis_dict())
|
|
||||||
|
|
||||||
# Set expiry
|
|
||||||
await client.expire(key, self._ttl_days * 24 * 60 * 60)
|
|
||||||
|
|
||||||
# Add to sorted set for time-based queries
|
|
||||||
index_key = f"benchmark_index:{benchmark.operation}"
|
|
||||||
await client.zadd(index_key, {key: timestamp_ms})
|
|
||||||
await client.expire(index_key, self._ttl_days * 24 * 60 * 60)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
"benchmark_recorded",
|
|
||||||
operation=benchmark.operation,
|
|
||||||
duration=benchmark.duration_seconds,
|
|
||||||
success=benchmark.success,
|
|
||||||
)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(
|
|
||||||
"benchmark_recording_failed",
|
|
||||||
error=str(e),
|
|
||||||
operation=benchmark.operation,
|
|
||||||
)
|
|
||||||
# Don't fail the request if benchmarking fails
|
|
||||||
|
|
||||||
async def query(
|
|
||||||
self,
|
|
||||||
operation: str,
|
|
||||||
start_time: Optional[datetime] = None,
|
|
||||||
end_time: Optional[datetime] = None,
|
|
||||||
limit: int = 100,
|
|
||||||
) -> list[PerformanceBenchmark]:
|
|
||||||
"""
|
|
||||||
Query benchmarks by operation and time range.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
operation: Operation name to filter by
|
|
||||||
start_time: Start of time range (inclusive)
|
|
||||||
end_time: End of time range (inclusive)
|
|
||||||
limit: Maximum number of results
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
List of benchmarks matching the query
|
|
||||||
|
|
||||||
Example:
|
|
||||||
>>> from datetime import timedelta
|
|
||||||
>>> now = datetime.now(timezone.utc)
|
|
||||||
>>> yesterday = now - timedelta(days=1)
|
|
||||||
>>> benchmarks = await store.query(
|
|
||||||
... "steward_analysis",
|
|
||||||
... start_time=yesterday,
|
|
||||||
... limit=50
|
|
||||||
... )
|
|
||||||
"""
|
|
||||||
if not config.ENABLE_BENCHMARKS:
|
|
||||||
return []
|
|
||||||
|
|
||||||
try:
|
|
||||||
client = await self._get_client()
|
|
||||||
index_key = f"benchmark_index:{operation}"
|
|
||||||
|
|
||||||
# Convert time range to timestamps
|
|
||||||
min_score = (
|
|
||||||
int(start_time.timestamp() * 1000)
|
|
||||||
if start_time
|
|
||||||
else "-inf"
|
|
||||||
)
|
|
||||||
max_score = (
|
|
||||||
int(end_time.timestamp() * 1000)
|
|
||||||
if end_time
|
|
||||||
else "+inf"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Query sorted set
|
|
||||||
keys = await client.zrevrangebyscore(
|
|
||||||
index_key,
|
|
||||||
max_score,
|
|
||||||
min_score,
|
|
||||||
start=0,
|
|
||||||
num=limit,
|
|
||||||
)
|
|
||||||
|
|
||||||
# Fetch benchmark data
|
|
||||||
benchmarks = []
|
|
||||||
for key in keys:
|
|
||||||
data = await client.hgetall(key)
|
|
||||||
if data:
|
|
||||||
benchmarks.append(PerformanceBenchmark.from_redis_dict(data))
|
|
||||||
|
|
||||||
return benchmarks
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(
|
|
||||||
"benchmark_query_failed",
|
|
||||||
error=str(e),
|
|
||||||
operation=operation,
|
|
||||||
)
|
|
||||||
return []
|
|
||||||
|
|
||||||
async def get_statistics(
|
|
||||||
self,
|
|
||||||
operation: str,
|
|
||||||
start_time: Optional[datetime] = None,
|
|
||||||
end_time: Optional[datetime] = None,
|
|
||||||
) -> dict[str, Any]:
|
|
||||||
"""
|
|
||||||
Get aggregate statistics for an operation.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
operation: Operation name
|
|
||||||
start_time: Start of time range
|
|
||||||
end_time: End of time range
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Dictionary with statistics (count, avg_duration, success_rate, etc.)
|
|
||||||
|
|
||||||
Example:
|
|
||||||
>>> stats = await store.get_statistics("steward_analysis")
|
|
||||||
>>> print(f"Average duration: {stats['avg_duration']}s")
|
|
||||||
>>> print(f"Success rate: {stats['success_rate']}%")
|
|
||||||
"""
|
|
||||||
benchmarks = await self.query(operation, start_time, end_time, limit=1000)
|
|
||||||
|
|
||||||
if not benchmarks:
|
|
||||||
return {
|
|
||||||
"count": 0,
|
|
||||||
"avg_duration": 0.0,
|
|
||||||
"min_duration": 0.0,
|
|
||||||
"max_duration": 0.0,
|
|
||||||
"success_rate": 0.0,
|
|
||||||
}
|
|
||||||
|
|
||||||
durations = [b.duration_seconds for b in benchmarks]
|
|
||||||
successes = sum(1 for b in benchmarks if b.success)
|
|
||||||
|
|
||||||
return {
|
|
||||||
"count": len(benchmarks),
|
|
||||||
"avg_duration": sum(durations) / len(durations),
|
|
||||||
"min_duration": min(durations),
|
|
||||||
"max_duration": max(durations),
|
|
||||||
"success_rate": (successes / len(benchmarks)) * 100,
|
|
||||||
"total_successes": successes,
|
|
||||||
"total_failures": len(benchmarks) - successes,
|
|
||||||
}
|
|
||||||
|
|
||||||
async def get_tool_accuracy(
|
|
||||||
self,
|
|
||||||
start_time: Optional[datetime] = None,
|
|
||||||
end_time: Optional[datetime] = None,
|
|
||||||
) -> dict[str, Any]:
|
|
||||||
"""
|
|
||||||
Analyze tool recommendation accuracy.
|
|
||||||
|
|
||||||
Compares recommended tools vs actually used tools to measure
|
|
||||||
Steward's recommendation precision.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
start_time: Start of time range
|
|
||||||
end_time: End of time range
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Dictionary with accuracy metrics
|
|
||||||
|
|
||||||
Example:
|
|
||||||
>>> accuracy = await store.get_tool_accuracy()
|
|
||||||
>>> print(f"Precision: {accuracy['precision']}%")
|
|
||||||
"""
|
|
||||||
tool_calls = await self.query("tool_call", start_time, end_time, limit=1000)
|
|
||||||
|
|
||||||
if not tool_calls:
|
|
||||||
return {
|
|
||||||
"total_calls": 0,
|
|
||||||
"recommended_and_used": 0,
|
|
||||||
"recommended_not_used": 0,
|
|
||||||
"not_recommended_but_used": 0,
|
|
||||||
"precision": 0.0,
|
|
||||||
}
|
|
||||||
|
|
||||||
recommended_and_used = sum(
|
|
||||||
1 for b in tool_calls
|
|
||||||
if b.was_recommended and b.was_actually_used
|
|
||||||
)
|
|
||||||
not_recommended_but_used = sum(
|
|
||||||
1 for b in tool_calls
|
|
||||||
if not b.was_recommended and b.was_actually_used
|
|
||||||
)
|
|
||||||
|
|
||||||
total_used = sum(1 for b in tool_calls if b.was_actually_used)
|
|
||||||
precision = (
|
|
||||||
(recommended_and_used / total_used * 100) if total_used > 0 else 0.0
|
|
||||||
)
|
|
||||||
|
|
||||||
return {
|
|
||||||
"total_calls": len(tool_calls),
|
|
||||||
"total_used": total_used,
|
|
||||||
"recommended_and_used": recommended_and_used,
|
|
||||||
"not_recommended_but_used": not_recommended_but_used,
|
|
||||||
"precision": precision,
|
|
||||||
}
|
|
||||||
|
|
||||||
async def close(self) -> None:
|
|
||||||
"""Close Redis connection."""
|
|
||||||
if self._client:
|
|
||||||
await self._client.aclose()
|
|
||||||
self._client = None
|
|
||||||
|
|
||||||
|
|
||||||
# Global benchmark store instance
|
|
||||||
_benchmark_store: Optional[BenchmarkStore] = None
|
|
||||||
|
|
||||||
|
|
||||||
def get_benchmark_store() -> BenchmarkStore:
|
|
||||||
"""
|
|
||||||
Get global benchmark store instance.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
BenchmarkStore instance
|
|
||||||
"""
|
|
||||||
global _benchmark_store
|
|
||||||
if _benchmark_store is None:
|
|
||||||
_benchmark_store = BenchmarkStore()
|
|
||||||
return _benchmark_store
|
|
||||||
+193
-46
@@ -2,15 +2,48 @@
|
|||||||
Global application configuration.
|
Global application configuration.
|
||||||
Following best practice of splitting config across domains.
|
Following best practice of splitting config across domains.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
from pydantic import Field, HttpUrl
|
from pydantic import Field, HttpUrl, model_validator
|
||||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||||
|
|
||||||
|
# Tenant isolation constants (see docs: tenant-based isolation, no separate
|
||||||
|
# test infrastructure). The production tenant owns real data in the shared
|
||||||
|
# services (Qdrant/Neo4j/Wiki.js/Redis); everything non-production must run
|
||||||
|
# under the reserved test tenant or an explicit test_-prefixed namespace.
|
||||||
|
PRODUCTION_TENANT = "jpmschweitzer"
|
||||||
|
TEST_TENANT = "llm_tester"
|
||||||
|
TEST_TENANT_PREFIX = "test_"
|
||||||
|
|
||||||
|
|
||||||
|
def _get_version_from_pyproject() -> str:
|
||||||
|
"""
|
||||||
|
Load version from pyproject.toml.
|
||||||
|
|
||||||
|
Falls back to "unknown" if file cannot be read.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Find pyproject.toml relative to this file
|
||||||
|
config_dir = Path(__file__).parent
|
||||||
|
pyproject_path = config_dir.parent.parent / "pyproject.toml"
|
||||||
|
|
||||||
|
if pyproject_path.exists():
|
||||||
|
content = pyproject_path.read_text()
|
||||||
|
for line in content.splitlines():
|
||||||
|
if line.strip().startswith("version"):
|
||||||
|
# Parse: version = "1.0.0"
|
||||||
|
return line.split("=", 1)[1].strip().strip('"').strip("'")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
class Environment(str, Enum):
|
class Environment(str, Enum):
|
||||||
"""Application environment."""
|
"""Application environment."""
|
||||||
|
|
||||||
DEVELOPMENT = "development"
|
DEVELOPMENT = "development"
|
||||||
PRODUCTION = "production"
|
PRODUCTION = "production"
|
||||||
TESTING = "testing"
|
TESTING = "testing"
|
||||||
@@ -19,91 +52,159 @@ class Environment(str, Enum):
|
|||||||
class Config(BaseSettings):
|
class Config(BaseSettings):
|
||||||
"""
|
"""
|
||||||
Global application configuration.
|
Global application configuration.
|
||||||
|
|
||||||
Loads from environment variables and .env file.
|
Loads from environment variables and .env file.
|
||||||
Domain-specific configs should be in their respective modules.
|
Domain-specific configs should be in their respective modules.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
model_config = SettingsConfigDict(
|
model_config = SettingsConfigDict(
|
||||||
env_file=".env",
|
env_file=".env",
|
||||||
env_file_encoding="utf-8",
|
env_file_encoding="utf-8",
|
||||||
case_sensitive=True,
|
case_sensitive=True,
|
||||||
extra="ignore",
|
extra="ignore",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Application
|
# Application
|
||||||
APP_NAME: str = "OpenAI-Compatible API"
|
APP_NAME: str = "OpenAI-Compatible API"
|
||||||
APP_VERSION: str = "0.2.5"
|
APP_VERSION: str = Field(default_factory=_get_version_from_pyproject)
|
||||||
ENVIRONMENT: Environment = Environment.DEVELOPMENT
|
ENVIRONMENT: Environment = Environment.DEVELOPMENT
|
||||||
DEBUG: bool = Field(default=False, description="Debug mode")
|
DEBUG: bool = Field(default=False, description="Debug mode")
|
||||||
|
|
||||||
# API Configuration
|
# API Configuration
|
||||||
API_HOST: str = Field(default="0.0.0.0", description="API host")
|
API_HOST: str = Field(default="0.0.0.0", description="API host")
|
||||||
API_PORT: int = Field(default=8000, description="API port")
|
API_PORT: int = Field(default=8000, description="API port")
|
||||||
API_PREFIX: str = Field(default="/v1", description="API route prefix")
|
API_PREFIX: str = Field(default="/v1", description="API route prefix")
|
||||||
|
|
||||||
# Ollama Configuration
|
# Anthropic Configuration (Claude - cloud fallback)
|
||||||
OLLAMA_HOST: HttpUrl = Field(
|
ANTHROPIC_API_KEY: str | None = Field(
|
||||||
default="http://localhost:11434",
|
default=None, description="Anthropic API key for the Claude fallback backend"
|
||||||
description="Ollama server URL"
|
|
||||||
)
|
)
|
||||||
OLLAMA_DEFAULT_MODEL: str = Field(
|
ANTHROPIC_MODEL: str = Field(
|
||||||
default="mistral-nemo:latest",
|
default="claude-sonnet-5", description="Claude model for the fallback backend"
|
||||||
description="Default Ollama model"
|
|
||||||
)
|
)
|
||||||
OLLAMA_TIMEOUT: int = Field(
|
PREFER_CLOUD_BACKEND: bool = Field(
|
||||||
default=120,
|
default=False, description="Prefer Claude over Ollama (default: local-first)"
|
||||||
description="Ollama request timeout in seconds"
|
)
|
||||||
|
|
||||||
|
# Ollama Configuration (local - primary backend)
|
||||||
|
OLLAMA_HOST: HttpUrl = Field(default="http://localhost:11434", description="Ollama server URL")
|
||||||
|
OLLAMA_DEFAULT_MODEL: str = Field(default="gemma4:e2b", description="Default Ollama model")
|
||||||
|
OLLAMA_TIMEOUT: int = Field(default=120, description="Ollama request timeout in seconds")
|
||||||
|
STEWARD_TIMEOUT: int = Field(
|
||||||
|
default=60, description="Steward analysis timeout in seconds (gemma4 needs ~35s warm)"
|
||||||
)
|
)
|
||||||
STREAM_TIMEOUT: int = Field(
|
STREAM_TIMEOUT: int = Field(
|
||||||
default=20,
|
default=20, description="Timeout for each streaming turn in seconds"
|
||||||
description="Timeout for each streaming turn in seconds"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# SearXNG Configuration
|
# SearXNG Configuration
|
||||||
SEARXNG_HOST: HttpUrl = Field(
|
SEARXNG_HOST: HttpUrl = Field(
|
||||||
default="http://localhost:8087",
|
default="http://searxng:8080",
|
||||||
description="SearXNG server URL"
|
description="SearXNG server URL (container name; internal port 8080)",
|
||||||
)
|
|
||||||
SEARXNG_TIMEOUT: int = Field(
|
|
||||||
default=30,
|
|
||||||
description="SearXNG request timeout in seconds"
|
|
||||||
)
|
)
|
||||||
|
SEARXNG_TIMEOUT: int = Field(default=30, description="SearXNG request timeout in seconds")
|
||||||
|
|
||||||
# Redis Configuration
|
# Redis Configuration
|
||||||
REDIS_HOST: str = Field(
|
REDIS_HOST: str = Field(default="localhost", description="Redis server host")
|
||||||
default="localhost",
|
REDIS_PORT: int = Field(default=6379, description="Redis server port")
|
||||||
description="Redis server host"
|
REDIS_TIMEOUT: int = Field(default=5, description="Redis connection timeout in seconds")
|
||||||
|
|
||||||
|
# Library-Desk Configuration (The Librarian backend)
|
||||||
|
LIBRARIAN_TIMEOUT: int = Field(
|
||||||
|
default=180, description="Total time budget for a librarian delegation in seconds"
|
||||||
)
|
)
|
||||||
REDIS_PORT: int = Field(
|
LIBRARY_DESK_HOST: HttpUrl = Field(
|
||||||
default=6379,
|
default="http://library-desk:8089",
|
||||||
description="Redis server port"
|
description="Library-Desk API URL (container name; internal port 8089)",
|
||||||
)
|
)
|
||||||
REDIS_DB: int = Field(
|
LIBRARY_DESK_API_KEY: str = Field(
|
||||||
default=1,
|
default="", description="API key for Library-Desk authentication"
|
||||||
description="Redis database number"
|
|
||||||
)
|
)
|
||||||
REDIS_TIMEOUT: int = Field(
|
LIBRARY_DESK_TIMEOUT: int = Field(
|
||||||
default=5,
|
default=60, description="Library-Desk request timeout in seconds"
|
||||||
description="Redis connection timeout in seconds"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Core-API Configuration (The Housekeeper backend)
|
||||||
|
CORE_API_HOST: HttpUrl = Field(
|
||||||
|
default="http://core-api:8083",
|
||||||
|
description="Core-API URL for Home Assistant integration (container name; internal port 8083)",
|
||||||
|
)
|
||||||
|
CORE_API_KEY: str = Field(default="", description="API key for Core-API authentication")
|
||||||
|
CORE_API_TIMEOUT: int = Field(default=30, description="Core-API request timeout in seconds")
|
||||||
|
|
||||||
|
# Qdrant Configuration (Memory vector storage)
|
||||||
|
QDRANT_HOST: str = Field(default="localhost", description="Qdrant server host")
|
||||||
|
QDRANT_PORT: int = Field(default=6333, description="Qdrant server port")
|
||||||
|
QDRANT_EMBEDDING_DIM: int = Field(
|
||||||
|
default=768, description="Embedding dimension (768 for nomic-embed-text)"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Ollama Embedding Configuration
|
||||||
|
OLLAMA_EMBEDDING_MODEL: str = Field(
|
||||||
|
default="nomic-embed-text", description="Ollama model for embeddings"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Redis Memory Database
|
||||||
|
REDIS_MEMORY_DB: int = Field(default=1, description="Redis database number for memory cache")
|
||||||
|
REDIS_MEMORY_TTL_HOURS: int = Field(default=24, description="TTL for session context in hours")
|
||||||
|
|
||||||
# Logging
|
# Logging
|
||||||
LOG_LEVEL: str = Field(default="INFO", description="Logging level")
|
LOG_LEVEL: str | None = Field(
|
||||||
ENABLE_BENCHMARKS: bool = Field(default=True, description="Enable performance benchmarking")
|
default=None, description="Logging level (auto-set based on environment if not specified)"
|
||||||
|
)
|
||||||
|
|
||||||
|
# User Configuration
|
||||||
|
DEFAULT_USER: str | None = Field(
|
||||||
|
default=None,
|
||||||
|
description="Default user for single-user setup (auto-set based on environment if not specified)",
|
||||||
|
)
|
||||||
|
|
||||||
# CORS
|
# CORS
|
||||||
CORS_ORIGINS: list[str] = Field(
|
CORS_ORIGINS: list[str] = Field(default=["*"], description="Allowed CORS origins")
|
||||||
default=["*"],
|
|
||||||
description="Allowed CORS origins"
|
|
||||||
)
|
|
||||||
CORS_ALLOW_CREDENTIALS: bool = True
|
CORS_ALLOW_CREDENTIALS: bool = True
|
||||||
CORS_ALLOW_METHODS: list[str] = ["*"]
|
CORS_ALLOW_METHODS: list[str] = ["*"]
|
||||||
CORS_ALLOW_HEADERS: list[str] = ["*"]
|
CORS_ALLOW_HEADERS: list[str] = ["*"]
|
||||||
|
|
||||||
|
@model_validator(mode="after")
|
||||||
|
def _refuse_production_tenant_outside_production(self) -> "Config":
|
||||||
|
"""
|
||||||
|
Refuse startup when a non-production environment is explicitly
|
||||||
|
configured with the production tenant.
|
||||||
|
|
||||||
|
This is the hard stop of the tenant isolation guard: a dev/test
|
||||||
|
instance must never be able to read or write the production
|
||||||
|
tenant's data in the shared services.
|
||||||
|
|
||||||
|
The comparison is on the sanitized form: namespaces are derived
|
||||||
|
through sanitize_user_id(), so variants like "JPMSchweitzer" or
|
||||||
|
"jpmschweitzer." collide with the production namespaces and are
|
||||||
|
refused just as loudly.
|
||||||
|
"""
|
||||||
|
from src.core.multi_tenancy import sanitize_user_id
|
||||||
|
|
||||||
|
if (
|
||||||
|
self.ENVIRONMENT != Environment.PRODUCTION
|
||||||
|
and self.DEFAULT_USER is not None
|
||||||
|
and sanitize_user_id(self.DEFAULT_USER) == sanitize_user_id(PRODUCTION_TENANT)
|
||||||
|
):
|
||||||
|
raise ValueError(
|
||||||
|
f"Refusing to start: ENVIRONMENT={self.ENVIRONMENT.value} is "
|
||||||
|
f"explicitly configured with the production tenant "
|
||||||
|
f"'{PRODUCTION_TENANT}'. Non-production environments must use "
|
||||||
|
f"'{TEST_TENANT}' or a '{TEST_TENANT_PREFIX}'-prefixed tenant. "
|
||||||
|
f"Unset DEFAULT_USER or set ENVIRONMENT=production."
|
||||||
|
)
|
||||||
|
return self
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def redis_url(self) -> str:
|
def redis_memory_url(self) -> str:
|
||||||
"""Construct Redis connection URL."""
|
"""Construct Redis connection URL for memory cache."""
|
||||||
return f"redis://{self.REDIS_HOST}:{self.REDIS_PORT}/{self.REDIS_DB}"
|
return f"redis://{self.REDIS_HOST}:{self.REDIS_PORT}/{self.REDIS_MEMORY_DB}"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def qdrant_url(self) -> str:
|
||||||
|
"""Construct Qdrant server URL."""
|
||||||
|
return f"http://{self.QDRANT_HOST}:{self.QDRANT_PORT}"
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def log_format(self) -> str:
|
def log_format(self) -> str:
|
||||||
@@ -115,12 +216,58 @@ class Config(BaseSettings):
|
|||||||
"""
|
"""
|
||||||
return "json" if self.ENVIRONMENT == Environment.PRODUCTION else "console"
|
return "json" if self.ENVIRONMENT == Environment.PRODUCTION else "console"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def effective_log_level(self) -> str:
|
||||||
|
"""
|
||||||
|
Get effective log level, auto-determining from environment if not set.
|
||||||
|
|
||||||
|
- development: DEBUG (maximum verbosity)
|
||||||
|
- production: WARNING (minimal noise)
|
||||||
|
- testing: INFO
|
||||||
|
"""
|
||||||
|
if self.LOG_LEVEL is not None:
|
||||||
|
return self.LOG_LEVEL
|
||||||
|
if self.ENVIRONMENT == Environment.DEVELOPMENT:
|
||||||
|
return "DEBUG"
|
||||||
|
if self.ENVIRONMENT == Environment.PRODUCTION:
|
||||||
|
return "WARNING"
|
||||||
|
return "INFO"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def effective_default_user(self) -> str:
|
||||||
|
"""
|
||||||
|
Get effective default user (tenant), enforcing tenant isolation.
|
||||||
|
|
||||||
|
- production: DEFAULT_USER if set, else the production tenant
|
||||||
|
- development/testing: FORCED to the reserved test tenant
|
||||||
|
("llm_tester") - the only accepted overrides are the test tenant
|
||||||
|
itself or a "test_"-prefixed namespace. Any other DEFAULT_USER
|
||||||
|
value is treated as misconfiguration and ignored.
|
||||||
|
"""
|
||||||
|
if self.ENVIRONMENT == Environment.PRODUCTION:
|
||||||
|
return self.DEFAULT_USER or PRODUCTION_TENANT
|
||||||
|
|
||||||
|
if self.DEFAULT_USER is not None and (
|
||||||
|
self.DEFAULT_USER == TEST_TENANT or self.DEFAULT_USER.startswith(TEST_TENANT_PREFIX)
|
||||||
|
):
|
||||||
|
return self.DEFAULT_USER
|
||||||
|
return TEST_TENANT
|
||||||
|
|
||||||
|
@property
|
||||||
|
def tenant_forced(self) -> bool:
|
||||||
|
"""Whether the tenant guard overrode a misconfigured DEFAULT_USER."""
|
||||||
|
return (
|
||||||
|
self.ENVIRONMENT != Environment.PRODUCTION
|
||||||
|
and self.DEFAULT_USER is not None
|
||||||
|
and self.effective_default_user != self.DEFAULT_USER
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@lru_cache
|
@lru_cache
|
||||||
def get_config() -> Config:
|
def get_config() -> Config:
|
||||||
"""
|
"""
|
||||||
Get cached configuration instance.
|
Get cached configuration instance.
|
||||||
|
|
||||||
Uses lru_cache to ensure config is loaded once and reused.
|
Uses lru_cache to ensure config is loaded once and reused.
|
||||||
"""
|
"""
|
||||||
return Config()
|
return Config()
|
||||||
|
|||||||
@@ -0,0 +1,174 @@
|
|||||||
|
"""
|
||||||
|
Request context using ContextVar for async-safe user/conversation tracking.
|
||||||
|
|
||||||
|
ContextVar provides task-local storage that automatically propagates through
|
||||||
|
async calls, eliminating the need to thread user identity through every function.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
# At request entry (router):
|
||||||
|
token = current_user.set(request.user or get_default_user())
|
||||||
|
try:
|
||||||
|
await service.process(request)
|
||||||
|
finally:
|
||||||
|
current_user.reset(token)
|
||||||
|
|
||||||
|
# Anywhere in the codebase:
|
||||||
|
from src.core.context import get_user
|
||||||
|
user = get_user() # Returns current request's user
|
||||||
|
"""
|
||||||
|
|
||||||
|
from contextvars import ContextVar
|
||||||
|
from types import TracebackType
|
||||||
|
|
||||||
|
|
||||||
|
def get_default_user() -> str:
|
||||||
|
"""
|
||||||
|
Get default user from config (environment-aware).
|
||||||
|
|
||||||
|
- development/testing: llm_tester (isolated test scope)
|
||||||
|
- production: jpmschweitzer (real user)
|
||||||
|
"""
|
||||||
|
# Import here to avoid circular dependency
|
||||||
|
from src.core.config import config
|
||||||
|
|
||||||
|
return config.effective_default_user
|
||||||
|
|
||||||
|
|
||||||
|
# Request-scoped context variables (async-safe, isolated per request)
|
||||||
|
# Note: ContextVar default is evaluated at definition, so we use a sentinel
|
||||||
|
# and resolve the real default in get_user()
|
||||||
|
_USER_NOT_SET = "__user_not_set__"
|
||||||
|
current_user: ContextVar[str] = ContextVar("current_user", default=_USER_NOT_SET)
|
||||||
|
current_conversation: ContextVar[str | None] = ContextVar("current_conversation", default=None)
|
||||||
|
|
||||||
|
|
||||||
|
def apply_tenant_guard(user: str) -> str:
|
||||||
|
"""
|
||||||
|
Enforce tenant isolation at request-context resolution.
|
||||||
|
|
||||||
|
In non-production environments the production tenant must never be
|
||||||
|
the effective user - a request that explicitly asks for it is forced
|
||||||
|
to the reserved test tenant instead (with a loud log line).
|
||||||
|
|
||||||
|
Comparison happens on the *sanitized* form of the user: every local
|
||||||
|
namespace (Qdrant collection, Redis key) is derived through
|
||||||
|
sanitize_user_id(), so any raw variant that collides with the
|
||||||
|
production tenant after sanitization ("JPMSchweitzer",
|
||||||
|
"jpmschweitzer.", " jpmschweitzer", ...) would otherwise resolve to
|
||||||
|
the production namespaces. Those variants are forced too.
|
||||||
|
"""
|
||||||
|
# Import here to avoid circular dependency
|
||||||
|
from src.core.config import PRODUCTION_TENANT, TEST_TENANT, Environment, config
|
||||||
|
from src.core.multi_tenancy import sanitize_user_id
|
||||||
|
|
||||||
|
if config.ENVIRONMENT != Environment.PRODUCTION and sanitize_user_id(user) == sanitize_user_id(
|
||||||
|
PRODUCTION_TENANT
|
||||||
|
):
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
get_logger(__name__).warning(
|
||||||
|
"tenant_guard_forced",
|
||||||
|
environment=config.ENVIRONMENT.value,
|
||||||
|
requested_tenant=user,
|
||||||
|
forced_tenant=TEST_TENANT,
|
||||||
|
)
|
||||||
|
return TEST_TENANT
|
||||||
|
return user
|
||||||
|
|
||||||
|
|
||||||
|
def get_user() -> str:
|
||||||
|
"""
|
||||||
|
Get current user from request context.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
User identifier for the current request.
|
||||||
|
Falls back to environment-aware default if not set.
|
||||||
|
In non-production environments the production tenant is never
|
||||||
|
returned - the tenant guard forces the reserved test tenant.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
user = get_user() # "llm_tester" (dev) or "jpmschweitzer" (prod)
|
||||||
|
"""
|
||||||
|
user = current_user.get()
|
||||||
|
if user == _USER_NOT_SET:
|
||||||
|
return get_default_user()
|
||||||
|
return apply_tenant_guard(user)
|
||||||
|
|
||||||
|
|
||||||
|
def get_conversation_id() -> str | None:
|
||||||
|
"""
|
||||||
|
Get current conversation ID from request context.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Conversation ID if set, None otherwise.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
conv_id = get_conversation_id() # "conv_abc123" or None
|
||||||
|
"""
|
||||||
|
return current_conversation.get()
|
||||||
|
|
||||||
|
|
||||||
|
class RequestContext:
|
||||||
|
"""
|
||||||
|
Context manager for setting request-scoped context.
|
||||||
|
|
||||||
|
Provides a cleaner alternative to manual token management.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
async with RequestContext(user="alice", conversation_id="conv_123"):
|
||||||
|
# All code here sees user="alice"
|
||||||
|
result = await some_service.process()
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
user: str | None = None,
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize request context.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier (defaults to environment-aware user if None)
|
||||||
|
conversation_id: Conversation ID (optional)
|
||||||
|
"""
|
||||||
|
self.user = user or get_default_user()
|
||||||
|
self.conversation_id = conversation_id
|
||||||
|
self._user_token = None
|
||||||
|
self._conv_token = None
|
||||||
|
|
||||||
|
async def __aenter__(self) -> "RequestContext":
|
||||||
|
"""Set context variables on entry."""
|
||||||
|
self._user_token = current_user.set(self.user)
|
||||||
|
self._conv_token = current_conversation.set(self.conversation_id)
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(
|
||||||
|
self,
|
||||||
|
exc_type: type[BaseException] | None,
|
||||||
|
exc_val: BaseException | None,
|
||||||
|
exc_tb: TracebackType | None,
|
||||||
|
) -> None:
|
||||||
|
"""Reset context variables on exit."""
|
||||||
|
if self._user_token is not None:
|
||||||
|
current_user.reset(self._user_token)
|
||||||
|
if self._conv_token is not None:
|
||||||
|
current_conversation.reset(self._conv_token)
|
||||||
|
|
||||||
|
def __enter__(self) -> "RequestContext":
|
||||||
|
"""Sync context manager entry (for non-async code)."""
|
||||||
|
self._user_token = current_user.set(self.user)
|
||||||
|
self._conv_token = current_conversation.set(self.conversation_id)
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(
|
||||||
|
self,
|
||||||
|
exc_type: type[BaseException] | None,
|
||||||
|
exc_val: BaseException | None,
|
||||||
|
exc_tb: TracebackType | None,
|
||||||
|
) -> None:
|
||||||
|
"""Sync context manager exit."""
|
||||||
|
if self._user_token is not None:
|
||||||
|
current_user.reset(self._user_token)
|
||||||
|
if self._conv_token is not None:
|
||||||
|
current_conversation.reset(self._conv_token)
|
||||||
@@ -0,0 +1,275 @@
|
|||||||
|
"""
|
||||||
|
Ollama client for embeddings generation.
|
||||||
|
|
||||||
|
Provides async embedding operations via Ollama API:
|
||||||
|
- Text embedding generation
|
||||||
|
- Batch embedding support
|
||||||
|
- Health checks
|
||||||
|
|
||||||
|
Adapted from library-desk patterns.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from types import TracebackType
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from .config import config
|
||||||
|
from .logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class OllamaEmbeddingClient:
|
||||||
|
"""
|
||||||
|
Ollama API client for embeddings.
|
||||||
|
|
||||||
|
Uses the Ollama embeddings endpoint to generate vector representations
|
||||||
|
of text using the nomic-embed-text model (768 dimensions).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
client = OllamaEmbeddingClient()
|
||||||
|
embedding = await client.embed("Hello world")
|
||||||
|
await client.close()
|
||||||
|
|
||||||
|
Or with context manager:
|
||||||
|
async with OllamaEmbeddingClient() as client:
|
||||||
|
embedding = await client.embed("Hello world")
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
base_url: str | None = None,
|
||||||
|
model: str | None = None,
|
||||||
|
timeout: float = 120.0,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize Ollama embedding client.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
base_url: Ollama server URL (defaults to config.OLLAMA_HOST)
|
||||||
|
model: Embedding model name (defaults to config.OLLAMA_EMBEDDING_MODEL)
|
||||||
|
timeout: Request timeout in seconds (embeddings can be slow)
|
||||||
|
"""
|
||||||
|
self.base_url = (base_url or str(config.OLLAMA_HOST)).rstrip("/")
|
||||||
|
self.model = model or config.OLLAMA_EMBEDDING_MODEL
|
||||||
|
self.embeddings_url = f"{self.base_url}/api/embeddings"
|
||||||
|
self.tags_url = f"{self.base_url}/api/tags"
|
||||||
|
self._client: httpx.AsyncClient | None = None
|
||||||
|
self._timeout = timeout
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ollama_embedding_client_initialized",
|
||||||
|
base_url=self.base_url,
|
||||||
|
model=self.model,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _get_client(self) -> httpx.AsyncClient:
|
||||||
|
"""Get or create HTTP client."""
|
||||||
|
if self._client is None:
|
||||||
|
self._client = httpx.AsyncClient(timeout=self._timeout)
|
||||||
|
return self._client
|
||||||
|
|
||||||
|
async def __aenter__(self) -> "OllamaEmbeddingClient":
|
||||||
|
"""Async context manager entry."""
|
||||||
|
await self._get_client()
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(
|
||||||
|
self,
|
||||||
|
exc_type: type[BaseException] | None,
|
||||||
|
exc_val: BaseException | None,
|
||||||
|
exc_tb: TracebackType | None,
|
||||||
|
) -> None:
|
||||||
|
"""Async context manager exit."""
|
||||||
|
await self.close()
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
"""Close HTTP client."""
|
||||||
|
if self._client is not None:
|
||||||
|
await self._client.aclose()
|
||||||
|
self._client = None
|
||||||
|
|
||||||
|
async def embed(self, text: str) -> list[float] | None:
|
||||||
|
"""
|
||||||
|
Generate embedding for single text.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text: Text to embed
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Embedding vector (768-dimensional for nomic-embed-text) or None on failure
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> embedding = await client.embed("Hello world")
|
||||||
|
>>> len(embedding)
|
||||||
|
768
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"model": self.model,
|
||||||
|
"prompt": text,
|
||||||
|
}
|
||||||
|
|
||||||
|
response = await client.post(self.embeddings_url, json=payload)
|
||||||
|
response.raise_for_status()
|
||||||
|
data = response.json()
|
||||||
|
|
||||||
|
embedding = data.get("embedding")
|
||||||
|
if not embedding:
|
||||||
|
logger.error("ollama_embed_no_embedding", response_data=data)
|
||||||
|
return None
|
||||||
|
|
||||||
|
return embedding
|
||||||
|
|
||||||
|
except httpx.HTTPStatusError as e:
|
||||||
|
logger.error(
|
||||||
|
"ollama_embed_http_error",
|
||||||
|
status_code=e.response.status_code,
|
||||||
|
detail=e.response.text,
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("ollama_embed_failed", error=str(e), exc_info=True)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def embed_batch(
|
||||||
|
self,
|
||||||
|
texts: list[str],
|
||||||
|
show_progress: bool = False,
|
||||||
|
) -> list[list[float] | None]:
|
||||||
|
"""
|
||||||
|
Generate embeddings for multiple texts.
|
||||||
|
|
||||||
|
Note: Ollama doesn't support native batch embeddings, so this
|
||||||
|
sequentially calls embed() for each text.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
texts: List of texts to embed
|
||||||
|
show_progress: Log progress for large batches
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of embedding vectors (same order as input)
|
||||||
|
None entries for texts that failed to embed
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> texts = ["Hello", "World", "Test"]
|
||||||
|
>>> embeddings = await client.embed_batch(texts)
|
||||||
|
>>> len(embeddings)
|
||||||
|
3
|
||||||
|
"""
|
||||||
|
embeddings = []
|
||||||
|
|
||||||
|
for i, text in enumerate(texts):
|
||||||
|
if show_progress and i % 10 == 0:
|
||||||
|
logger.info(
|
||||||
|
"ollama_embed_batch_progress",
|
||||||
|
current=i,
|
||||||
|
total=len(texts),
|
||||||
|
)
|
||||||
|
|
||||||
|
embedding = await self.embed(text)
|
||||||
|
embeddings.append(embedding)
|
||||||
|
|
||||||
|
if show_progress:
|
||||||
|
logger.info(
|
||||||
|
"ollama_embed_batch_complete",
|
||||||
|
successful=sum(1 for e in embeddings if e is not None),
|
||||||
|
total=len(texts),
|
||||||
|
)
|
||||||
|
|
||||||
|
return embeddings
|
||||||
|
|
||||||
|
async def embed_batch_filtered(
|
||||||
|
self,
|
||||||
|
texts: list[str],
|
||||||
|
show_progress: bool = False,
|
||||||
|
) -> list[list[float]]:
|
||||||
|
"""
|
||||||
|
Generate embeddings for multiple texts, filtering out failures.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
texts: List of texts to embed
|
||||||
|
show_progress: Log progress for large batches
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of successful embedding vectors (may be shorter than input)
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> embeddings = await client.embed_batch_filtered(texts)
|
||||||
|
>>> all(e is not None for e in embeddings)
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
all_embeddings = await self.embed_batch(texts, show_progress)
|
||||||
|
return [e for e in all_embeddings if e is not None]
|
||||||
|
|
||||||
|
async def get_embedding_dimension(self) -> int | None:
|
||||||
|
"""
|
||||||
|
Get embedding dimension for current model.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Embedding dimension (e.g., 768 for nomic-embed-text) or None on failure
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> dim = await client.get_embedding_dimension()
|
||||||
|
>>> dim
|
||||||
|
768
|
||||||
|
"""
|
||||||
|
test_embedding = await self.embed("test")
|
||||||
|
if test_embedding:
|
||||||
|
return len(test_embedding)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def health_check(self) -> bool:
|
||||||
|
"""
|
||||||
|
Check if Ollama server is reachable and model is available.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if healthy, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
response = await client.get(self.tags_url, timeout=5.0)
|
||||||
|
response.raise_for_status()
|
||||||
|
data = response.json()
|
||||||
|
models = data.get("models", [])
|
||||||
|
|
||||||
|
# Check if our embedding model is available
|
||||||
|
model_found = False
|
||||||
|
for m in models:
|
||||||
|
name = m.get("name", "")
|
||||||
|
if name == self.model or name.startswith(f"{self.model}:"):
|
||||||
|
model_found = True
|
||||||
|
break
|
||||||
|
|
||||||
|
if not model_found:
|
||||||
|
logger.warning(
|
||||||
|
"ollama_embedding_model_not_found",
|
||||||
|
model=self.model,
|
||||||
|
available=[m.get("name") for m in models],
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("ollama_embedding_health_check_failed", error=str(e))
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# Global client instance (lazy initialization)
|
||||||
|
_embedding_client: OllamaEmbeddingClient | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def get_embedding_client() -> OllamaEmbeddingClient:
|
||||||
|
"""
|
||||||
|
Get global embedding client instance.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
OllamaEmbeddingClient instance
|
||||||
|
"""
|
||||||
|
global _embedding_client
|
||||||
|
if _embedding_client is None:
|
||||||
|
_embedding_client = OllamaEmbeddingClient()
|
||||||
|
return _embedding_client
|
||||||
@@ -2,12 +2,13 @@
|
|||||||
Global exception definitions.
|
Global exception definitions.
|
||||||
Domain-specific exceptions should be in their respective modules.
|
Domain-specific exceptions should be in their respective modules.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
class AppException(Exception):
|
class AppException(Exception):
|
||||||
"""Base exception for all application errors."""
|
"""Base exception for all application errors."""
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
message: str = "An error occurred",
|
message: str = "An error occurred",
|
||||||
@@ -22,26 +23,26 @@ class AppException(Exception):
|
|||||||
|
|
||||||
class OllamaConnectionError(AppException):
|
class OllamaConnectionError(AppException):
|
||||||
"""Raised when cannot connect to Ollama service."""
|
"""Raised when cannot connect to Ollama service."""
|
||||||
|
|
||||||
def __init__(self, message: str = "Cannot connect to Ollama service"):
|
def __init__(self, message: str = "Cannot connect to Ollama service"):
|
||||||
super().__init__(message=message, status_code=503)
|
super().__init__(message=message, status_code=503)
|
||||||
|
|
||||||
|
|
||||||
class OllamaTimeoutError(AppException):
|
class OllamaTimeoutError(AppException):
|
||||||
"""Raised when Ollama request times out."""
|
"""Raised when Ollama request times out."""
|
||||||
|
|
||||||
def __init__(self, message: str = "Ollama request timed out"):
|
def __init__(self, message: str = "Ollama request timed out"):
|
||||||
super().__init__(message=message, status_code=504)
|
super().__init__(message=message, status_code=504)
|
||||||
|
|
||||||
|
|
||||||
class ModelNotFoundError(AppException):
|
class ModelNotFoundError(AppException):
|
||||||
"""Raised when requested model is not available."""
|
"""Raised when requested model is not available."""
|
||||||
|
|
||||||
def __init__(self, model_name: str):
|
def __init__(self, model_name: str):
|
||||||
super().__init__(
|
super().__init__(
|
||||||
message=f"Model '{model_name}' not found",
|
message=f"Model '{model_name}' not found",
|
||||||
status_code=404,
|
status_code=404,
|
||||||
details={"model": model_name}
|
details={"model": model_name},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -5,10 +5,10 @@ Provides centralized registry of household members (agents) with their
|
|||||||
capabilities and tools. Supports two-tier abstraction: executive summaries
|
capabilities and tools. Supports two-tier abstraction: executive summaries
|
||||||
for coordination and full toolsets for execution.
|
for coordination and full toolsets for execution.
|
||||||
"""
|
"""
|
||||||
from typing import Any, Optional
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from pydantic import BaseModel, ConfigDict
|
from pydantic import BaseModel, ConfigDict
|
||||||
from pydantic_ai import Agent
|
|
||||||
|
|
||||||
from .logging_config import get_logger
|
from .logging_config import get_logger
|
||||||
|
|
||||||
@@ -22,6 +22,7 @@ class HouseholdCapability(BaseModel):
|
|||||||
This is what the Steward and Butler see for coordination.
|
This is what the Steward and Butler see for coordination.
|
||||||
High-level description without implementation details.
|
High-level description without implementation details.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
name: str # Unique identifier: "tatlock_core", "librarian", "developer"
|
name: str # Unique identifier: "tatlock_core", "librarian", "developer"
|
||||||
role: str # Display name: "Butler's Core Tools", "The Librarian"
|
role: str # Display name: "Butler's Core Tools", "The Librarian"
|
||||||
category: str # "core", "research", "technical", "automation"
|
category: str # "core", "research", "technical", "automation"
|
||||||
@@ -38,11 +39,12 @@ class HouseholdMember(BaseModel):
|
|||||||
Contains both the executive summary (for coordination) and
|
Contains both the executive summary (for coordination) and
|
||||||
implementation details (tools/agent).
|
implementation details (tools/agent).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||||
|
|
||||||
capability: HouseholdCapability
|
capability: HouseholdCapability
|
||||||
tools: list[Any] # PydanticAI tool definitions (any type since Tool is a dataclass)
|
tools: list[Any] # PydanticAI tool definitions (any type since Tool is a dataclass)
|
||||||
agent: Optional[Any] = None # For expert agents (Phase 4)
|
agent: Any | None = None # For expert agents (Phase 4)
|
||||||
|
|
||||||
|
|
||||||
class HouseholdRegistry:
|
class HouseholdRegistry:
|
||||||
@@ -55,7 +57,7 @@ class HouseholdRegistry:
|
|||||||
3. Agent delegation (Phase 4)
|
3. Agent delegation (Phase 4)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self):
|
def __init__(self) -> None:
|
||||||
"""Initialize empty registry."""
|
"""Initialize empty registry."""
|
||||||
self._members: dict[str, HouseholdMember] = {}
|
self._members: dict[str, HouseholdMember] = {}
|
||||||
logger.info("household_registry_initialized")
|
logger.info("household_registry_initialized")
|
||||||
@@ -65,7 +67,7 @@ class HouseholdRegistry:
|
|||||||
name: str,
|
name: str,
|
||||||
capability: HouseholdCapability,
|
capability: HouseholdCapability,
|
||||||
tools: list[Any],
|
tools: list[Any],
|
||||||
agent: Optional[Any] = None,
|
agent: Any | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Register a household member.
|
Register a household member.
|
||||||
@@ -95,9 +97,7 @@ class HouseholdRegistry:
|
|||||||
... )
|
... )
|
||||||
"""
|
"""
|
||||||
if name != capability.name:
|
if name != capability.name:
|
||||||
raise ValueError(
|
raise ValueError(f"Name mismatch: '{name}' != '{capability.name}'")
|
||||||
f"Name mismatch: '{name}' != '{capability.name}'"
|
|
||||||
)
|
|
||||||
|
|
||||||
self._members[name] = HouseholdMember(
|
self._members[name] = HouseholdMember(
|
||||||
capability=capability,
|
capability=capability,
|
||||||
@@ -132,7 +132,7 @@ class HouseholdRegistry:
|
|||||||
role=member.capability.role,
|
role=member.capability.role,
|
||||||
)
|
)
|
||||||
|
|
||||||
def get_member(self, name: str) -> Optional[HouseholdMember]:
|
def get_member(self, name: str) -> HouseholdMember | None:
|
||||||
"""
|
"""
|
||||||
Get full household member specification.
|
Get full household member specification.
|
||||||
|
|
||||||
@@ -200,6 +200,79 @@ class HouseholdRegistry:
|
|||||||
|
|
||||||
return tools
|
return tools
|
||||||
|
|
||||||
|
def get_delegation_tools(self, names: list[str]) -> list[Any]:
|
||||||
|
"""
|
||||||
|
Get delegation wrapper tools for specified capabilities.
|
||||||
|
|
||||||
|
Instead of returning raw tools (which overloads the LLM),
|
||||||
|
returns wrapper functions that delegate to expert agents.
|
||||||
|
This implements the agent-as-tool pattern.
|
||||||
|
|
||||||
|
For members WITH an agent: returns delegation wrapper
|
||||||
|
For members WITHOUT an agent (e.g., tatlock_core): returns raw tools
|
||||||
|
|
||||||
|
Args:
|
||||||
|
names: List of member names to include
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of delegation wrappers and/or raw tools
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> # Steward recommends librarian + tatlock_core
|
||||||
|
>>> tools = registry.get_delegation_tools(["librarian", "tatlock_core"])
|
||||||
|
>>> # Returns: [delegate_to_librarian, calculate, datetime, ...]
|
||||||
|
>>> # Instead of: [hybrid_search, search_wiki, create_wiki_page, ... (16 tools)]
|
||||||
|
"""
|
||||||
|
from src.agents.delegation import (
|
||||||
|
delegate_to_biographer,
|
||||||
|
delegate_to_housekeeper,
|
||||||
|
delegate_to_librarian,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Map of expert names to their delegation wrappers
|
||||||
|
delegation_wrappers = {
|
||||||
|
"librarian": delegate_to_librarian,
|
||||||
|
"biographer": delegate_to_biographer,
|
||||||
|
"housekeeper": delegate_to_housekeeper,
|
||||||
|
}
|
||||||
|
|
||||||
|
tools = []
|
||||||
|
for name in names:
|
||||||
|
member = self._members.get(name)
|
||||||
|
if not member:
|
||||||
|
logger.warning(
|
||||||
|
"household_member_not_found",
|
||||||
|
requested_name=name,
|
||||||
|
available_names=list(self._members.keys()),
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Check if this member has a delegation wrapper
|
||||||
|
if name in delegation_wrappers and member.agent is not None:
|
||||||
|
# Use delegation wrapper instead of raw tools
|
||||||
|
tools.append(delegation_wrappers[name])
|
||||||
|
logger.debug(
|
||||||
|
"delegation_wrapper_added",
|
||||||
|
member=name,
|
||||||
|
wrapper=delegation_wrappers[name].__name__,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# No agent = direct tools (e.g., tatlock_core)
|
||||||
|
tools.extend(member.tools)
|
||||||
|
logger.debug(
|
||||||
|
"raw_tools_added",
|
||||||
|
member=name,
|
||||||
|
tool_count=len(member.tools),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"delegation_tools_created",
|
||||||
|
requested_members=names,
|
||||||
|
total_tools=len(tools),
|
||||||
|
)
|
||||||
|
|
||||||
|
return tools
|
||||||
|
|
||||||
def list_members(self) -> list[str]:
|
def list_members(self) -> list[str]:
|
||||||
"""
|
"""
|
||||||
List all registered member names.
|
List all registered member names.
|
||||||
|
|||||||
+38
-18
@@ -4,12 +4,14 @@ Structured logging configuration using structlog.
|
|||||||
Deeply integrates with FastAPI/uvicorn's built-in logging to provide
|
Deeply integrates with FastAPI/uvicorn's built-in logging to provide
|
||||||
seamless structured logs across the entire application stack.
|
seamless structured logs across the entire application stack.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.config
|
import logging.config
|
||||||
import sys
|
import sys
|
||||||
|
from collections.abc import AsyncIterator
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
from datetime import datetime, timezone
|
from datetime import UTC, datetime
|
||||||
from typing import Any, AsyncIterator
|
from typing import Any
|
||||||
|
|
||||||
import structlog
|
import structlog
|
||||||
from structlog.types import EventDict, Processor
|
from structlog.types import EventDict, Processor
|
||||||
@@ -19,7 +21,7 @@ from .config import config
|
|||||||
|
|
||||||
def add_timestamp(logger: Any, method_name: str, event_dict: EventDict) -> EventDict:
|
def add_timestamp(logger: Any, method_name: str, event_dict: EventDict) -> EventDict:
|
||||||
"""Add ISO 8601 timestamp to log entries."""
|
"""Add ISO 8601 timestamp to log entries."""
|
||||||
event_dict["timestamp"] = datetime.now(timezone.utc).isoformat()
|
event_dict["timestamp"] = datetime.now(UTC).isoformat()
|
||||||
return event_dict
|
return event_dict
|
||||||
|
|
||||||
|
|
||||||
@@ -41,11 +43,28 @@ def extract_from_record(logger: Any, method_name: str, event_dict: EventDict) ->
|
|||||||
# Extract custom fields from record
|
# Extract custom fields from record
|
||||||
for key, value in record.__dict__.items():
|
for key, value in record.__dict__.items():
|
||||||
if key not in {
|
if key not in {
|
||||||
"name", "msg", "args", "created", "filename", "funcName",
|
"name",
|
||||||
"levelname", "levelno", "lineno", "module", "msecs",
|
"msg",
|
||||||
"message", "pathname", "process", "processName", "relativeCreated",
|
"args",
|
||||||
"thread", "threadName", "exc_info", "exc_text", "stack_info",
|
"created",
|
||||||
"taskName"
|
"filename",
|
||||||
|
"funcName",
|
||||||
|
"levelname",
|
||||||
|
"levelno",
|
||||||
|
"lineno",
|
||||||
|
"module",
|
||||||
|
"msecs",
|
||||||
|
"message",
|
||||||
|
"pathname",
|
||||||
|
"process",
|
||||||
|
"processName",
|
||||||
|
"relativeCreated",
|
||||||
|
"thread",
|
||||||
|
"threadName",
|
||||||
|
"exc_info",
|
||||||
|
"exc_text",
|
||||||
|
"stack_info",
|
||||||
|
"taskName",
|
||||||
}:
|
}:
|
||||||
event_dict[key] = value
|
event_dict[key] = value
|
||||||
|
|
||||||
@@ -122,7 +141,7 @@ def configure_logging() -> None:
|
|||||||
root_logger = logging.getLogger()
|
root_logger = logging.getLogger()
|
||||||
root_logger.handlers.clear()
|
root_logger.handlers.clear()
|
||||||
root_logger.addHandler(handler)
|
root_logger.addHandler(handler)
|
||||||
root_logger.setLevel(logging.getLevelName(config.LOG_LEVEL))
|
root_logger.setLevel(logging.getLevelName(config.effective_log_level))
|
||||||
|
|
||||||
# Configure specific loggers
|
# Configure specific loggers
|
||||||
for logger_name in [
|
for logger_name in [
|
||||||
@@ -135,7 +154,7 @@ def configure_logging() -> None:
|
|||||||
logger = logging.getLogger(logger_name)
|
logger = logging.getLogger(logger_name)
|
||||||
logger.handlers.clear()
|
logger.handlers.clear()
|
||||||
logger.propagate = True
|
logger.propagate = True
|
||||||
logger.setLevel(logging.getLevelName(config.LOG_LEVEL))
|
logger.setLevel(logging.getLevelName(config.effective_log_level))
|
||||||
|
|
||||||
|
|
||||||
def get_logger(name: str) -> structlog.stdlib.BoundLogger:
|
def get_logger(name: str) -> structlog.stdlib.BoundLogger:
|
||||||
@@ -164,7 +183,7 @@ def get_logger(name: str) -> structlog.stdlib.BoundLogger:
|
|||||||
async def log_operation(
|
async def log_operation(
|
||||||
operation: str,
|
operation: str,
|
||||||
initial_context: dict[str, Any] | None = None,
|
initial_context: dict[str, Any] | None = None,
|
||||||
logger_name: str = "tatlock.operations"
|
logger_name: str = "tatlock.operations",
|
||||||
) -> AsyncIterator[dict[str, Any]]:
|
) -> AsyncIterator[dict[str, Any]]:
|
||||||
"""
|
"""
|
||||||
Context manager for automatic operation timing and logging.
|
Context manager for automatic operation timing and logging.
|
||||||
@@ -187,21 +206,21 @@ async def log_operation(
|
|||||||
context = initial_context or {}
|
context = initial_context or {}
|
||||||
context["operation"] = operation
|
context["operation"] = operation
|
||||||
|
|
||||||
start_time = datetime.now(timezone.utc)
|
start_time = datetime.now(UTC)
|
||||||
logger.info("operation_started", **context)
|
logger.info("operation_started", **context)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
yield context
|
yield context
|
||||||
|
|
||||||
# Success case
|
# Success case
|
||||||
duration = (datetime.now(timezone.utc) - start_time).total_seconds()
|
duration = (datetime.now(UTC) - start_time).total_seconds()
|
||||||
context["duration_seconds"] = duration
|
context["duration_seconds"] = duration
|
||||||
context["success"] = True
|
context["success"] = True
|
||||||
logger.info("operation_completed", **context)
|
logger.info("operation_completed", **context)
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
# Error case
|
# Error case
|
||||||
duration = (datetime.now(timezone.utc) - start_time).total_seconds()
|
duration = (datetime.now(UTC) - start_time).total_seconds()
|
||||||
context["duration_seconds"] = duration
|
context["duration_seconds"] = duration
|
||||||
context["success"] = False
|
context["success"] = False
|
||||||
context["error"] = str(e)
|
context["error"] = str(e)
|
||||||
@@ -228,7 +247,8 @@ def get_uvicorn_log_config() -> dict[str, Any]:
|
|||||||
"()": structlog.stdlib.ProcessorFormatter,
|
"()": structlog.stdlib.ProcessorFormatter,
|
||||||
"processors": [
|
"processors": [
|
||||||
structlog.stdlib.ProcessorFormatter.remove_processors_meta,
|
structlog.stdlib.ProcessorFormatter.remove_processors_meta,
|
||||||
structlog.processors.JSONRenderer() if config.log_format == "json"
|
structlog.processors.JSONRenderer()
|
||||||
|
if config.log_format == "json"
|
||||||
else structlog.dev.ConsoleRenderer(colors=True),
|
else structlog.dev.ConsoleRenderer(colors=True),
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
@@ -241,9 +261,9 @@ def get_uvicorn_log_config() -> dict[str, Any]:
|
|||||||
},
|
},
|
||||||
},
|
},
|
||||||
"loggers": {
|
"loggers": {
|
||||||
"uvicorn": {"handlers": ["default"], "level": config.LOG_LEVEL},
|
"uvicorn": {"handlers": ["default"], "level": config.effective_log_level},
|
||||||
"uvicorn.error": {"handlers": ["default"], "level": config.LOG_LEVEL},
|
"uvicorn.error": {"handlers": ["default"], "level": config.effective_log_level},
|
||||||
"uvicorn.access": {"handlers": ["default"], "level": config.LOG_LEVEL},
|
"uvicorn.access": {"handlers": ["default"], "level": config.effective_log_level},
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,391 @@
|
|||||||
|
"""
|
||||||
|
Redis-backed memory cache for session context.
|
||||||
|
|
||||||
|
Provides short-term memory storage with TTL:
|
||||||
|
- Session context (24h TTL)
|
||||||
|
- Recent entities mentioned in conversation
|
||||||
|
- User-scoped with conversation isolation
|
||||||
|
|
||||||
|
Uses Redis DB 1.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import redis.asyncio as redis
|
||||||
|
|
||||||
|
from .config import config
|
||||||
|
from .logging_config import get_logger
|
||||||
|
from .multi_tenancy import get_entities_key, get_session_key
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class MemoryCache:
|
||||||
|
"""
|
||||||
|
Redis-backed cache for session memory.
|
||||||
|
|
||||||
|
Stores ephemeral context that doesn't need vector search:
|
||||||
|
- Session context (recent topics, user state)
|
||||||
|
- Recent entities (people, places, things mentioned)
|
||||||
|
- Conversation metadata
|
||||||
|
|
||||||
|
All data expires after REDIS_MEMORY_TTL_HOURS (default 24h).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
cache = MemoryCache()
|
||||||
|
await cache.set_session_context(
|
||||||
|
user="jpmschweitzer",
|
||||||
|
conversation_id="conv_123",
|
||||||
|
context={"topic": "docker", "mood": "curious"}
|
||||||
|
)
|
||||||
|
context = await cache.get_session_context("jpmschweitzer", "conv_123")
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
redis_url: str | None = None,
|
||||||
|
ttl_hours: int | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize memory cache.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
redis_url: Redis connection URL (defaults to config.redis_memory_url)
|
||||||
|
ttl_hours: TTL for cached data (defaults to config.REDIS_MEMORY_TTL_HOURS)
|
||||||
|
"""
|
||||||
|
self._redis_url = redis_url or config.redis_memory_url
|
||||||
|
self._ttl_seconds = (ttl_hours or config.REDIS_MEMORY_TTL_HOURS) * 3600
|
||||||
|
self._client: redis.Redis | None = None
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"memory_cache_initialized",
|
||||||
|
redis_url=self._redis_url,
|
||||||
|
ttl_hours=ttl_hours or config.REDIS_MEMORY_TTL_HOURS,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _get_client(self) -> redis.Redis:
|
||||||
|
"""Get or create Redis client."""
|
||||||
|
if self._client is None:
|
||||||
|
self._client = redis.from_url(
|
||||||
|
self._redis_url,
|
||||||
|
encoding="utf-8",
|
||||||
|
decode_responses=True,
|
||||||
|
socket_timeout=config.REDIS_TIMEOUT,
|
||||||
|
socket_connect_timeout=config.REDIS_TIMEOUT,
|
||||||
|
)
|
||||||
|
return self._client
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
"""Close Redis connection."""
|
||||||
|
if self._client is not None:
|
||||||
|
await self._client.aclose()
|
||||||
|
self._client = None
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Session Context
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def get_session_context(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
) -> dict[str, Any] | None:
|
||||||
|
"""
|
||||||
|
Get session context for a conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Session context dict or None if not found
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> context = await cache.get_session_context("jpmschweitzer", "conv_123")
|
||||||
|
>>> context
|
||||||
|
{"topic": "docker", "mood": "curious", "last_tool": "librarian"}
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_session_key(user, conversation_id)
|
||||||
|
|
||||||
|
data = await client.get(key)
|
||||||
|
if data is None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
return json.loads(data)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_get_session_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def set_session_context(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
context: dict[str, Any],
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Set session context for a conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
context: Context data to store
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful, False otherwise
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await cache.set_session_context(
|
||||||
|
... "jpmschweitzer",
|
||||||
|
... "conv_123",
|
||||||
|
... {"topic": "docker", "mood": "curious"}
|
||||||
|
... )
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_session_key(user, conversation_id)
|
||||||
|
|
||||||
|
await client.setex(
|
||||||
|
key,
|
||||||
|
self._ttl_seconds,
|
||||||
|
json.dumps(context),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"memory_cache_set_session",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
context_keys=list(context.keys()),
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_set_session_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def update_session_context(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
updates: dict[str, Any],
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Update session context (merge with existing).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
updates: Fields to update/add
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
existing = await self.get_session_context(user, conversation_id) or {}
|
||||||
|
existing.update(updates)
|
||||||
|
return await self.set_session_context(user, conversation_id, existing)
|
||||||
|
|
||||||
|
async def delete_session_context(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Delete session context for a conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if deleted, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_session_key(user, conversation_id)
|
||||||
|
await client.delete(key)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_delete_session_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Recent Entities
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def get_recent_entities(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
) -> list[str]:
|
||||||
|
"""
|
||||||
|
Get recently mentioned entities in a conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of entity names/identifiers
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> entities = await cache.get_recent_entities("jpmschweitzer", "conv_123")
|
||||||
|
>>> entities
|
||||||
|
["Docker", "Kubernetes", "nginx"]
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_entities_key(user, conversation_id)
|
||||||
|
|
||||||
|
# Get all members of the set
|
||||||
|
entities = await client.smembers(key)
|
||||||
|
return list(entities)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_get_entities_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
async def add_recent_entities(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
entities: list[str],
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Add entities to the recent entities set.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
entities: Entity names to add
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful, False otherwise
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await cache.add_recent_entities(
|
||||||
|
... "jpmschweitzer",
|
||||||
|
... "conv_123",
|
||||||
|
... ["Docker", "Kubernetes"]
|
||||||
|
... )
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
if not entities:
|
||||||
|
return True
|
||||||
|
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_entities_key(user, conversation_id)
|
||||||
|
|
||||||
|
# Add to set
|
||||||
|
await client.sadd(key, *entities)
|
||||||
|
|
||||||
|
# Refresh TTL
|
||||||
|
await client.expire(key, self._ttl_seconds)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"memory_cache_add_entities",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
entities=entities,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_add_entities_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def clear_recent_entities(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
conversation_id: str,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Clear all recent entities for a conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if cleared, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
key = get_entities_key(user, conversation_id)
|
||||||
|
await client.delete(key)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_cache_clear_entities_failed",
|
||||||
|
user=user,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Health Check
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def health_check(self) -> bool:
|
||||||
|
"""
|
||||||
|
Check if Redis is reachable.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if healthy, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
client = await self._get_client()
|
||||||
|
await client.ping()
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("memory_cache_health_check_failed", error=str(e))
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# Global cache instance (lazy initialization)
|
||||||
|
_memory_cache: MemoryCache | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def get_memory_cache() -> MemoryCache:
|
||||||
|
"""
|
||||||
|
Get global memory cache instance.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
MemoryCache instance
|
||||||
|
"""
|
||||||
|
global _memory_cache
|
||||||
|
if _memory_cache is None:
|
||||||
|
_memory_cache = MemoryCache()
|
||||||
|
return _memory_cache
|
||||||
@@ -0,0 +1,621 @@
|
|||||||
|
"""
|
||||||
|
Memory service for direct key-based access.
|
||||||
|
|
||||||
|
Provides fast, LLM-free access to user memories for:
|
||||||
|
- Known-key lookups (location, timezone, preferences)
|
||||||
|
- Session context (current topic, recent entities)
|
||||||
|
- Structured storage (explicit user instructions)
|
||||||
|
|
||||||
|
This is the "direct access layer" - no LLM interpretation.
|
||||||
|
For semantic/fuzzy queries, use the Memory Agent instead.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
from src.core.memory_service import memory_service
|
||||||
|
|
||||||
|
# Get user's location (fast, no LLM)
|
||||||
|
location = await memory_service.get_profile("location")
|
||||||
|
|
||||||
|
# Set a preference
|
||||||
|
await memory_service.set_preference("temperature_unit", "celsius")
|
||||||
|
|
||||||
|
# Get session context
|
||||||
|
ctx = await memory_service.get_session_context(conversation_id)
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from enum import Enum
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
|
from .context import get_conversation_id, get_user
|
||||||
|
from .embeddings import get_embedding_client
|
||||||
|
from .logging_config import get_logger
|
||||||
|
from .memory_cache import get_memory_cache
|
||||||
|
from .multi_tenancy import get_memory_collection_name
|
||||||
|
from .qdrant import get_qdrant_client
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class MemoryType(str, Enum):
|
||||||
|
"""Types of memories stored in Qdrant."""
|
||||||
|
|
||||||
|
USER_PROFILE = "user_profile" # Name, location, timezone
|
||||||
|
PREFERENCE = "preference" # Units, language, theme
|
||||||
|
LEARNED_FACT = "learned_fact" # "My car is a Tesla"
|
||||||
|
|
||||||
|
|
||||||
|
class MemoryRecord(BaseModel):
|
||||||
|
"""A memory record stored in Qdrant."""
|
||||||
|
|
||||||
|
id: str
|
||||||
|
type: MemoryType
|
||||||
|
key: str # e.g., "location", "timezone", "car"
|
||||||
|
value: str # The actual content
|
||||||
|
keywords: list[str] = Field(default_factory=list)
|
||||||
|
importance: float = 0.5 # 0.0 - 1.0
|
||||||
|
source: str = "explicit" # "explicit" | "inferred" | "conversation"
|
||||||
|
created_at: str = Field(default_factory=lambda: datetime.now(UTC).isoformat())
|
||||||
|
updated_at: str = Field(default_factory=lambda: datetime.now(UTC).isoformat())
|
||||||
|
|
||||||
|
|
||||||
|
class MemoryService:
|
||||||
|
"""
|
||||||
|
Direct access to user memories without LLM overhead.
|
||||||
|
|
||||||
|
Use this for:
|
||||||
|
- Known-key lookups: get_profile("location"), get_preference("units")
|
||||||
|
- Explicit storage: set_preference("theme", "dark")
|
||||||
|
- Session context: get_session_context(), update_session_context()
|
||||||
|
|
||||||
|
Do NOT use for:
|
||||||
|
- Fuzzy queries: "What car do I drive?" → Use Memory Agent
|
||||||
|
- Semantic recall: "What did I mention about X?" → Use Memory Agent
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
"""Initialize memory service with lazy client loading."""
|
||||||
|
self._qdrant = None
|
||||||
|
self._embedding = None
|
||||||
|
self._cache = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def qdrant(self):
|
||||||
|
"""Lazy-load Qdrant client."""
|
||||||
|
if self._qdrant is None:
|
||||||
|
self._qdrant = get_qdrant_client()
|
||||||
|
return self._qdrant
|
||||||
|
|
||||||
|
@property
|
||||||
|
def embedding(self):
|
||||||
|
"""Lazy-load embedding client."""
|
||||||
|
if self._embedding is None:
|
||||||
|
self._embedding = get_embedding_client()
|
||||||
|
return self._embedding
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cache(self):
|
||||||
|
"""Lazy-load Redis cache."""
|
||||||
|
if self._cache is None:
|
||||||
|
self._cache = get_memory_cache()
|
||||||
|
return self._cache
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Profile Methods (user_profile type)
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def get_profile(self, key: str, user: str | None = None) -> str | None:
|
||||||
|
"""
|
||||||
|
Get a user profile value by key.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Profile key (e.g., "location", "timezone", "name")
|
||||||
|
user: User ID (defaults to current request context)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Profile value or None if not found
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> location = await memory_service.get_profile("location")
|
||||||
|
>>> location
|
||||||
|
"Amsterdam, Netherlands"
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._get_memory(user, MemoryType.USER_PROFILE, key)
|
||||||
|
|
||||||
|
async def set_profile(
|
||||||
|
self,
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
user: str | None = None,
|
||||||
|
keywords: list[str] | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Set a user profile value.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Profile key (e.g., "location", "timezone")
|
||||||
|
value: Profile value
|
||||||
|
user: User ID (defaults to current request context)
|
||||||
|
keywords: Optional keywords for semantic search
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await memory_service.set_profile("location", "Amsterdam, Netherlands")
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._set_memory(
|
||||||
|
user=user,
|
||||||
|
memory_type=MemoryType.USER_PROFILE,
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
keywords=keywords or [key],
|
||||||
|
importance=0.9, # Profile data is important
|
||||||
|
)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Preference Methods (preference type)
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def get_preference(self, key: str, user: str | None = None) -> str | None:
|
||||||
|
"""
|
||||||
|
Get a user preference by key.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Preference key (e.g., "temperature_unit", "language", "theme")
|
||||||
|
user: User ID (defaults to current request context)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Preference value or None if not found
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> units = await memory_service.get_preference("temperature_unit")
|
||||||
|
>>> units
|
||||||
|
"celsius"
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._get_memory(user, MemoryType.PREFERENCE, key)
|
||||||
|
|
||||||
|
async def set_preference(
|
||||||
|
self,
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Set a user preference.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Preference key
|
||||||
|
value: Preference value
|
||||||
|
user: User ID (defaults to current request context)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await memory_service.set_preference("theme", "dark")
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._set_memory(
|
||||||
|
user=user,
|
||||||
|
memory_type=MemoryType.PREFERENCE,
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
keywords=[key, "preference"],
|
||||||
|
importance=0.7,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def get_all_preferences(self, user: str | None = None) -> dict[str, str]:
|
||||||
|
"""
|
||||||
|
Get all preferences for a user.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict of key -> value for all preferences
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
memories = await self._get_all_by_type(user, MemoryType.PREFERENCE)
|
||||||
|
return {m["key"]: m["value"] for m in memories}
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Learned Facts (learned_fact type) - for direct storage only
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def store_fact(
|
||||||
|
self,
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
user: str | None = None,
|
||||||
|
keywords: list[str] | None = None,
|
||||||
|
importance: float = 0.5,
|
||||||
|
source: str = "explicit",
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Store a learned fact about the user.
|
||||||
|
|
||||||
|
Use this for explicit user statements like:
|
||||||
|
- "Remember that my car is a Tesla"
|
||||||
|
- "I work at Acme Corp"
|
||||||
|
|
||||||
|
For semantic extraction from conversation, use the Memory Agent.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Fact identifier (e.g., "car", "employer")
|
||||||
|
value: The fact content
|
||||||
|
user: User ID
|
||||||
|
keywords: Keywords for semantic search
|
||||||
|
importance: 0.0-1.0 importance score
|
||||||
|
source: "explicit" | "inferred" | "conversation"
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._set_memory(
|
||||||
|
user=user,
|
||||||
|
memory_type=MemoryType.LEARNED_FACT,
|
||||||
|
key=key,
|
||||||
|
value=value,
|
||||||
|
keywords=keywords or [key],
|
||||||
|
importance=importance,
|
||||||
|
source=source,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def get_fact(self, key: str, user: str | None = None) -> str | None:
|
||||||
|
"""
|
||||||
|
Get a specific fact by key.
|
||||||
|
|
||||||
|
For semantic/fuzzy queries, use the Memory Agent.
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
return await self._get_memory(user, MemoryType.LEARNED_FACT, key)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Session Context (Redis-backed, 24h TTL)
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def get_session_context(
|
||||||
|
self,
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> dict[str, Any] | None:
|
||||||
|
"""
|
||||||
|
Get session context for current conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
conversation_id: Conversation ID (defaults to current context)
|
||||||
|
user: User ID (defaults to current context)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Session context dict or None
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
conversation_id = conversation_id or get_conversation_id()
|
||||||
|
|
||||||
|
if not conversation_id:
|
||||||
|
return None
|
||||||
|
|
||||||
|
return await self.cache.get_session_context(user, conversation_id)
|
||||||
|
|
||||||
|
async def set_session_context(
|
||||||
|
self,
|
||||||
|
context: dict[str, Any],
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Set session context for current conversation.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
context: Context data to store
|
||||||
|
conversation_id: Conversation ID
|
||||||
|
user: User ID
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
conversation_id = conversation_id or get_conversation_id()
|
||||||
|
|
||||||
|
if not conversation_id:
|
||||||
|
logger.warning("memory_service_no_conversation_id")
|
||||||
|
return False
|
||||||
|
|
||||||
|
return await self.cache.set_session_context(user, conversation_id, context)
|
||||||
|
|
||||||
|
async def update_session_context(
|
||||||
|
self,
|
||||||
|
updates: dict[str, Any],
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Update session context (merge with existing).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
updates: Fields to update
|
||||||
|
conversation_id: Conversation ID
|
||||||
|
user: User ID
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
conversation_id = conversation_id or get_conversation_id()
|
||||||
|
|
||||||
|
if not conversation_id:
|
||||||
|
return False
|
||||||
|
|
||||||
|
return await self.cache.update_session_context(user, conversation_id, updates)
|
||||||
|
|
||||||
|
async def get_recent_entities(
|
||||||
|
self,
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> list[str]:
|
||||||
|
"""
|
||||||
|
Get recently mentioned entities in conversation.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of entity names
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
conversation_id = conversation_id or get_conversation_id()
|
||||||
|
|
||||||
|
if not conversation_id:
|
||||||
|
return []
|
||||||
|
|
||||||
|
return await self.cache.get_recent_entities(user, conversation_id)
|
||||||
|
|
||||||
|
async def add_recent_entities(
|
||||||
|
self,
|
||||||
|
entities: list[str],
|
||||||
|
conversation_id: str | None = None,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Add entities to recent entities set.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entities: Entity names to add
|
||||||
|
conversation_id: Conversation ID
|
||||||
|
user: User ID
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
conversation_id = conversation_id or get_conversation_id()
|
||||||
|
|
||||||
|
if not conversation_id:
|
||||||
|
return False
|
||||||
|
|
||||||
|
return await self.cache.add_recent_entities(user, conversation_id, entities)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Bulk / Pre-fetch Methods (for Steward)
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def prefetch_context(
|
||||||
|
self,
|
||||||
|
user: str | None = None,
|
||||||
|
include_profile: bool = True,
|
||||||
|
include_preferences: bool = True,
|
||||||
|
profile_keys: list[str] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""
|
||||||
|
Pre-fetch commonly needed context for Steward.
|
||||||
|
|
||||||
|
This is the main entry point for Steward to get user context
|
||||||
|
before analyzing a request.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User ID
|
||||||
|
include_profile: Include profile data
|
||||||
|
include_preferences: Include preferences
|
||||||
|
profile_keys: Specific profile keys to fetch (None = common ones)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with profile and preferences data
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> ctx = await memory_service.prefetch_context()
|
||||||
|
>>> ctx
|
||||||
|
{
|
||||||
|
"profile": {"location": "Amsterdam", "timezone": "Europe/Amsterdam"},
|
||||||
|
"preferences": {"temperature_unit": "celsius"}
|
||||||
|
}
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
result: dict[str, Any] = {}
|
||||||
|
|
||||||
|
if include_profile:
|
||||||
|
profile_keys = profile_keys or ["location", "timezone", "name"]
|
||||||
|
profile = {}
|
||||||
|
for key in profile_keys:
|
||||||
|
value = await self.get_profile(key, user)
|
||||||
|
if value:
|
||||||
|
profile[key] = value
|
||||||
|
if profile:
|
||||||
|
result["profile"] = profile
|
||||||
|
|
||||||
|
if include_preferences:
|
||||||
|
preferences = await self.get_all_preferences(user)
|
||||||
|
if preferences:
|
||||||
|
result["preferences"] = preferences
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"memory_service_prefetch",
|
||||||
|
user=user,
|
||||||
|
profile_keys=list(result.get("profile", {}).keys()),
|
||||||
|
preference_keys=list(result.get("preferences", {}).keys()),
|
||||||
|
)
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Internal Methods
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
async def _get_memory(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
memory_type: MemoryType,
|
||||||
|
key: str,
|
||||||
|
) -> str | None:
|
||||||
|
"""Get a memory by type and key (exact match)."""
|
||||||
|
collection = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Search with filter for exact type + key match
|
||||||
|
# We use a dummy vector since we're filtering by payload
|
||||||
|
results = self.qdrant._client.scroll(
|
||||||
|
collection_name=collection,
|
||||||
|
scroll_filter={
|
||||||
|
"must": [
|
||||||
|
{"key": "type", "match": {"value": memory_type.value}},
|
||||||
|
{"key": "key", "match": {"value": key}},
|
||||||
|
]
|
||||||
|
},
|
||||||
|
limit=1,
|
||||||
|
with_payload=True,
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
points, _ = results
|
||||||
|
if points:
|
||||||
|
return points[0].payload.get("value")
|
||||||
|
return None
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_service_get_failed",
|
||||||
|
user=user,
|
||||||
|
type=memory_type.value,
|
||||||
|
key=key,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def _set_memory(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
memory_type: MemoryType,
|
||||||
|
key: str,
|
||||||
|
value: str,
|
||||||
|
keywords: list[str],
|
||||||
|
importance: float = 0.5,
|
||||||
|
source: str = "explicit",
|
||||||
|
) -> bool:
|
||||||
|
"""Set a memory (upsert by type + key)."""
|
||||||
|
try:
|
||||||
|
# Generate embedding for semantic search
|
||||||
|
embedding = await self.embedding.embed(f"{key}: {value}")
|
||||||
|
if not embedding:
|
||||||
|
logger.error("memory_service_embedding_failed", key=key)
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Create memory ID from type + key for idempotent upserts
|
||||||
|
memory_id = f"{memory_type.value}:{key}"
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"type": memory_type.value,
|
||||||
|
"key": key,
|
||||||
|
"value": value,
|
||||||
|
"keywords": keywords,
|
||||||
|
"importance": importance,
|
||||||
|
"source": source,
|
||||||
|
"updated_at": datetime.now(UTC).isoformat(),
|
||||||
|
}
|
||||||
|
|
||||||
|
result = await self.qdrant.upsert_memory(
|
||||||
|
user=user,
|
||||||
|
memory_id=memory_id,
|
||||||
|
vector=embedding,
|
||||||
|
payload=payload,
|
||||||
|
)
|
||||||
|
|
||||||
|
if result:
|
||||||
|
logger.debug(
|
||||||
|
"memory_service_set",
|
||||||
|
user=user,
|
||||||
|
type=memory_type.value,
|
||||||
|
key=key,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"memory_service_set_failed",
|
||||||
|
user=user,
|
||||||
|
type=memory_type.value,
|
||||||
|
key=key,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def _get_all_by_type(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
memory_type: MemoryType,
|
||||||
|
limit: int = 100,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Get all memories of a specific type."""
|
||||||
|
collection = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
results = self.qdrant._client.scroll(
|
||||||
|
collection_name=collection,
|
||||||
|
scroll_filter={
|
||||||
|
"must": [
|
||||||
|
{"key": "type", "match": {"value": memory_type.value}},
|
||||||
|
]
|
||||||
|
},
|
||||||
|
limit=limit,
|
||||||
|
with_payload=True,
|
||||||
|
with_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
points, _ = results
|
||||||
|
return [p.payload for p in points]
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"memory_service_get_all_failed",
|
||||||
|
user=user,
|
||||||
|
type=memory_type.value,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
async def delete_memory(
|
||||||
|
self,
|
||||||
|
key: str,
|
||||||
|
memory_type: MemoryType,
|
||||||
|
user: str | None = None,
|
||||||
|
) -> bool:
|
||||||
|
"""
|
||||||
|
Delete a specific memory.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
key: Memory key
|
||||||
|
memory_type: Type of memory
|
||||||
|
user: User ID
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if deleted
|
||||||
|
"""
|
||||||
|
user = user or get_user()
|
||||||
|
memory_id = f"{memory_type.value}:{key}"
|
||||||
|
|
||||||
|
return await self.qdrant.delete_memory(user, memory_id)
|
||||||
|
|
||||||
|
|
||||||
|
# Global service instance
|
||||||
|
memory_service = MemoryService()
|
||||||
+6
-5
@@ -2,6 +2,7 @@
|
|||||||
Custom Pydantic base models for consistent serialization.
|
Custom Pydantic base models for consistent serialization.
|
||||||
Following best practice of having a global base model.
|
Following best practice of having a global base model.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
@@ -17,12 +18,13 @@ def datetime_to_iso_str(dt: datetime) -> str:
|
|||||||
class CustomBaseModel(BaseModel):
|
class CustomBaseModel(BaseModel):
|
||||||
"""
|
"""
|
||||||
Custom base model with consistent configuration.
|
Custom base model with consistent configuration.
|
||||||
|
|
||||||
All domain models should inherit from this for:
|
All domain models should inherit from this for:
|
||||||
- Consistent JSON serialization
|
- Consistent JSON serialization
|
||||||
- Timezone-aware datetime handling
|
- Timezone-aware datetime handling
|
||||||
- Alias population support
|
- Alias population support
|
||||||
"""
|
"""
|
||||||
|
|
||||||
model_config = ConfigDict(
|
model_config = ConfigDict(
|
||||||
json_encoders={datetime: datetime_to_iso_str},
|
json_encoders={datetime: datetime_to_iso_str},
|
||||||
populate_by_name=True,
|
populate_by_name=True,
|
||||||
@@ -30,14 +32,13 @@ class CustomBaseModel(BaseModel):
|
|||||||
validate_assignment=True,
|
validate_assignment=True,
|
||||||
arbitrary_types_allowed=True,
|
arbitrary_types_allowed=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
def serializable_dict(self, **kwargs: Any) -> dict[str, Any]:
|
def serializable_dict(self, **kwargs: Any) -> dict[str, Any]:
|
||||||
"""
|
"""
|
||||||
Return dict with only JSON-serializable fields.
|
Return dict with only JSON-serializable fields.
|
||||||
|
|
||||||
Useful for logging and debugging.
|
Useful for logging and debugging.
|
||||||
"""
|
"""
|
||||||
return jsonable_encoder(
|
return jsonable_encoder(
|
||||||
self.model_dump(**kwargs),
|
self.model_dump(**kwargs), custom_encoder={datetime: datetime_to_iso_str}
|
||||||
custom_encoder={datetime: datetime_to_iso_str}
|
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,148 @@
|
|||||||
|
"""
|
||||||
|
Multi-tenancy helpers for Tatlock.
|
||||||
|
|
||||||
|
Provides utilities for user namespace management across:
|
||||||
|
- Qdrant (collection per user for memories)
|
||||||
|
- Redis (user-scoped keys for session context)
|
||||||
|
|
||||||
|
Adapted from library-desk patterns.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
|
||||||
|
def sanitize_user_id(user_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Sanitize user ID for use in collection names, keys, and paths.
|
||||||
|
|
||||||
|
Converts special characters to underscores and ensures alphanumeric safety.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_id: Raw user identifier (email, username, etc.)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Sanitized user ID safe for use in identifiers
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
>>> sanitize_user_id("john@example.com")
|
||||||
|
'john_at_example_com'
|
||||||
|
>>> sanitize_user_id("user.name")
|
||||||
|
'user_name'
|
||||||
|
>>> sanitize_user_id("User Name")
|
||||||
|
'user_name'
|
||||||
|
"""
|
||||||
|
sanitized = user_id.lower()
|
||||||
|
|
||||||
|
# Convert @ to _at_
|
||||||
|
sanitized = sanitized.replace("@", "_at_")
|
||||||
|
|
||||||
|
# Convert dots to underscores
|
||||||
|
sanitized = sanitized.replace(".", "_")
|
||||||
|
|
||||||
|
# Replace any non-alphanumeric characters with underscores
|
||||||
|
sanitized = re.sub(r"[^a-z0-9_]", "_", sanitized)
|
||||||
|
|
||||||
|
# Remove consecutive underscores
|
||||||
|
sanitized = re.sub(r"_+", "_", sanitized)
|
||||||
|
|
||||||
|
# Remove leading/trailing underscores
|
||||||
|
sanitized = sanitized.strip("_")
|
||||||
|
|
||||||
|
return sanitized
|
||||||
|
|
||||||
|
|
||||||
|
def get_memory_collection_name(user_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Get Qdrant collection name for user's memories.
|
||||||
|
|
||||||
|
Pattern: memories_{sanitized_user_id}
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_id: User identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Qdrant collection name
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
>>> get_memory_collection_name("jpmschweitzer")
|
||||||
|
'memories_jpmschweitzer'
|
||||||
|
>>> get_memory_collection_name("john@example.com")
|
||||||
|
'memories_john_at_example_com'
|
||||||
|
"""
|
||||||
|
sanitized = sanitize_user_id(user_id)
|
||||||
|
return f"memories_{sanitized}"
|
||||||
|
|
||||||
|
|
||||||
|
def get_session_key(user_id: str, conversation_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Get Redis key for session context.
|
||||||
|
|
||||||
|
Pattern: session:{sanitized_user}:{conversation_id}
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_id: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Redis key for session context
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
>>> get_session_key("jpmschweitzer", "conv_abc123")
|
||||||
|
'session:jpmschweitzer:conv_abc123'
|
||||||
|
"""
|
||||||
|
sanitized = sanitize_user_id(user_id)
|
||||||
|
return f"session:{sanitized}:{conversation_id}"
|
||||||
|
|
||||||
|
|
||||||
|
def get_entities_key(user_id: str, conversation_id: str) -> str:
|
||||||
|
"""
|
||||||
|
Get Redis key for recent entities in a conversation.
|
||||||
|
|
||||||
|
Pattern: entities:{sanitized_user}:{conversation_id}
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_id: User identifier
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Redis key for recent entities
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
>>> get_entities_key("jpmschweitzer", "conv_abc123")
|
||||||
|
'entities:jpmschweitzer:conv_abc123'
|
||||||
|
"""
|
||||||
|
sanitized = sanitize_user_id(user_id)
|
||||||
|
return f"entities:{sanitized}:{conversation_id}"
|
||||||
|
|
||||||
|
|
||||||
|
def validate_user_id(user_id: str) -> bool:
|
||||||
|
"""
|
||||||
|
Validate that a user ID is acceptable.
|
||||||
|
|
||||||
|
Checks:
|
||||||
|
- Not empty
|
||||||
|
- Not too long (max 100 chars)
|
||||||
|
- Contains some alphanumeric characters
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user_id: User identifier to validate
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if valid, False otherwise
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
>>> validate_user_id("jpmschweitzer")
|
||||||
|
True
|
||||||
|
>>> validate_user_id("")
|
||||||
|
False
|
||||||
|
>>> validate_user_id("a" * 101)
|
||||||
|
False
|
||||||
|
"""
|
||||||
|
if not user_id or len(user_id) > 100:
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Must contain at least one alphanumeric character
|
||||||
|
if not re.search(r"[a-zA-Z0-9]", user_id):
|
||||||
|
return False
|
||||||
|
|
||||||
|
return True
|
||||||
+59
-13
@@ -3,17 +3,37 @@ Request preprocessing pipeline.
|
|||||||
|
|
||||||
Analyzes requests via the Steward and creates scoped toolsets for Tatlock.
|
Analyzes requests via the Steward and creates scoped toolsets for Tatlock.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Any, Optional
|
from datetime import datetime
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from src.agents.steward import analyze_request, format_steward_note
|
from src.agents.steward import analyze_request, format_steward_note
|
||||||
from src.agents.steward.schemas import StewardRecommendation
|
from src.agents.steward.schemas import StewardRecommendation
|
||||||
from src.core.household_registry import get_household_registry
|
from src.core.household_registry import get_household_registry
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
|
from src.core.tracing import SpanType, trace_span
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _inject_temporal_context(request: str) -> str:
|
||||||
|
"""
|
||||||
|
Append current time context to user request.
|
||||||
|
|
||||||
|
Provides Tatlock with temporal awareness for time-sensitive queries.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
request: Original user request
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Request with appended time context
|
||||||
|
"""
|
||||||
|
now = datetime.now()
|
||||||
|
time_str = now.strftime("%Y-%m-%d %H:%M")
|
||||||
|
return f"{request}\n\n[Current time: {time_str}]"
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class EnrichedRequest:
|
class EnrichedRequest:
|
||||||
"""
|
"""
|
||||||
@@ -26,6 +46,7 @@ class EnrichedRequest:
|
|||||||
recommendation: Full Steward recommendation
|
recommendation: Full Steward recommendation
|
||||||
steward_reasoning: Plain text reasoning for streaming to user
|
steward_reasoning: Plain text reasoning for streaming to user
|
||||||
"""
|
"""
|
||||||
|
|
||||||
original_request: str
|
original_request: str
|
||||||
steward_note: str
|
steward_note: str
|
||||||
scoped_tools: list[Any] # PydanticAI tool definitions
|
scoped_tools: list[Any] # PydanticAI tool definitions
|
||||||
@@ -36,7 +57,7 @@ class EnrichedRequest:
|
|||||||
async def preprocess_request(
|
async def preprocess_request(
|
||||||
user_request: str,
|
user_request: str,
|
||||||
conversation_history: list[dict],
|
conversation_history: list[dict],
|
||||||
conversation_id: Optional[str] = None,
|
conversation_id: str | None = None,
|
||||||
) -> EnrichedRequest:
|
) -> EnrichedRequest:
|
||||||
"""
|
"""
|
||||||
Analyze request via Steward and prepare scoped context for Tatlock.
|
Analyze request via Steward and prepare scoped context for Tatlock.
|
||||||
@@ -65,6 +86,9 @@ async def preprocess_request(
|
|||||||
>>> print(len(enriched.scoped_tools))
|
>>> print(len(enriched.scoped_tools))
|
||||||
5 # All tatlock_core tools
|
5 # All tatlock_core tools
|
||||||
"""
|
"""
|
||||||
|
# Inject temporal context for time-aware processing
|
||||||
|
enriched_request = _inject_temporal_context(user_request)
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"preprocessing_request",
|
"preprocessing_request",
|
||||||
request_preview=user_request[:100],
|
request_preview=user_request[:100],
|
||||||
@@ -72,21 +96,43 @@ async def preprocess_request(
|
|||||||
conversation_id=conversation_id,
|
conversation_id=conversation_id,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Call Steward with full conversation history
|
# Call Steward with full conversation history (traced)
|
||||||
recommendation = await analyze_request(
|
async with trace_span(
|
||||||
user_request,
|
"steward_analysis",
|
||||||
conversation_history=conversation_history,
|
SpanType.STEWARD,
|
||||||
conversation_id=conversation_id,
|
metadata={
|
||||||
)
|
"request_preview": user_request[:100],
|
||||||
|
"history_length": len(conversation_history),
|
||||||
|
},
|
||||||
|
) as span:
|
||||||
|
recommendation = await analyze_request(
|
||||||
|
enriched_request,
|
||||||
|
conversation_history=conversation_history,
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Update span with results
|
||||||
|
if span:
|
||||||
|
span.metadata.update(
|
||||||
|
{
|
||||||
|
"recommended_capabilities": recommendation.recommended_capabilities,
|
||||||
|
"complexity": recommendation.estimated_complexity,
|
||||||
|
"has_memory_context": bool(recommendation.memory_context),
|
||||||
|
"has_conversation_context": recommendation.conversation_context.has_previous_context,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
span.details["reasoning"] = recommendation.reasoning
|
||||||
|
if recommendation.enriched_query:
|
||||||
|
span.details["enriched_query"] = recommendation.enriched_query
|
||||||
|
|
||||||
# Format note for Tatlock (includes conversation context)
|
# Format note for Tatlock (includes conversation context)
|
||||||
steward_note = await format_steward_note(recommendation)
|
steward_note = await format_steward_note(recommendation)
|
||||||
|
|
||||||
# Get scoped tools from household registry
|
# Get delegation tools from household registry
|
||||||
|
# Uses agent-as-tool pattern: expert agents get delegation wrappers,
|
||||||
|
# core tools are returned directly
|
||||||
registry = get_household_registry()
|
registry = get_household_registry()
|
||||||
scoped_tools = registry.get_scoped_tools(
|
scoped_tools = registry.get_delegation_tools(recommendation.recommended_capabilities)
|
||||||
recommendation.recommended_capabilities
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"preprocessing_complete",
|
"preprocessing_complete",
|
||||||
@@ -97,7 +143,7 @@ async def preprocess_request(
|
|||||||
)
|
)
|
||||||
|
|
||||||
return EnrichedRequest(
|
return EnrichedRequest(
|
||||||
original_request=user_request,
|
original_request=enriched_request,
|
||||||
steward_note=steward_note,
|
steward_note=steward_note,
|
||||||
scoped_tools=scoped_tools,
|
scoped_tools=scoped_tools,
|
||||||
recommendation=recommendation,
|
recommendation=recommendation,
|
||||||
|
|||||||
@@ -0,0 +1,460 @@
|
|||||||
|
"""
|
||||||
|
Qdrant client wrapper for memory vector storage.
|
||||||
|
|
||||||
|
Provides async operations for storing and retrieving memory embeddings:
|
||||||
|
- Collection management (per-user collections)
|
||||||
|
- Memory upsert/search/delete
|
||||||
|
- Filtering by memory type
|
||||||
|
|
||||||
|
Adapted from library-desk patterns.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
from uuid import NAMESPACE_DNS, uuid4, uuid5
|
||||||
|
|
||||||
|
from qdrant_client import QdrantClient
|
||||||
|
from qdrant_client.http import models as qdrant_models
|
||||||
|
|
||||||
|
from .config import config
|
||||||
|
from .logging_config import get_logger
|
||||||
|
from .multi_tenancy import get_memory_collection_name
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class MemoryQdrantClient:
|
||||||
|
"""
|
||||||
|
Qdrant client wrapper for memory storage.
|
||||||
|
|
||||||
|
Manages per-user collections with the pattern: memories_{user}
|
||||||
|
Stores memory embeddings with metadata (type, content, timestamps).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
client = MemoryQdrantClient()
|
||||||
|
await client.ensure_collection("jpmschweitzer")
|
||||||
|
await client.upsert_memory(
|
||||||
|
user="jpmschweitzer",
|
||||||
|
memory_id="mem_123",
|
||||||
|
vector=[0.1, 0.2, ...],
|
||||||
|
payload={"type": "fact", "content": "User prefers dark mode"}
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
url: str | None = None,
|
||||||
|
embedding_dim: int | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize Qdrant client.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
url: Qdrant server URL (defaults to config.qdrant_url)
|
||||||
|
embedding_dim: Vector dimension (defaults to config.QDRANT_EMBEDDING_DIM)
|
||||||
|
"""
|
||||||
|
self.url = url or config.qdrant_url
|
||||||
|
self.embedding_dim = embedding_dim or config.QDRANT_EMBEDDING_DIM
|
||||||
|
self._client = QdrantClient(url=self.url)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"qdrant_client_initialized",
|
||||||
|
url=self.url,
|
||||||
|
embedding_dim=self.embedding_dim,
|
||||||
|
)
|
||||||
|
|
||||||
|
def close(self) -> None:
|
||||||
|
"""Close Qdrant client."""
|
||||||
|
if self._client is not None:
|
||||||
|
self._client.close()
|
||||||
|
|
||||||
|
async def ensure_collection(self, user: str) -> bool:
|
||||||
|
"""
|
||||||
|
Ensure collection exists for user, create if not.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if collection exists or was created successfully
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await client.ensure_collection("jpmschweitzer")
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Check if collection exists
|
||||||
|
collections = self._client.get_collections()
|
||||||
|
existing = [c.name for c in collections.collections]
|
||||||
|
|
||||||
|
if collection_name in existing:
|
||||||
|
logger.debug(
|
||||||
|
"qdrant_collection_exists",
|
||||||
|
collection=collection_name,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Create collection with cosine distance
|
||||||
|
self._client.create_collection(
|
||||||
|
collection_name=collection_name,
|
||||||
|
vectors_config=qdrant_models.VectorParams(
|
||||||
|
size=self.embedding_dim,
|
||||||
|
distance=qdrant_models.Distance.COSINE,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"qdrant_collection_created",
|
||||||
|
collection=collection_name,
|
||||||
|
embedding_dim=self.embedding_dim,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_ensure_collection_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def upsert_memory(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
memory_id: str | None,
|
||||||
|
vector: list[float],
|
||||||
|
payload: dict[str, Any],
|
||||||
|
) -> str | None:
|
||||||
|
"""
|
||||||
|
Upsert a memory point.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
memory_id: Memory ID (generated if None)
|
||||||
|
vector: Embedding vector
|
||||||
|
payload: Memory metadata (should include 'type', 'content', etc.)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Memory ID if successful, None on failure
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> memory_id = await client.upsert_memory(
|
||||||
|
... user="jpmschweitzer",
|
||||||
|
... memory_id=None,
|
||||||
|
... vector=[0.1, 0.2, ...],
|
||||||
|
... payload={
|
||||||
|
... "type": "fact",
|
||||||
|
... "content": "User prefers dark mode",
|
||||||
|
... "created_at": "2024-01-01T00:00:00Z"
|
||||||
|
... }
|
||||||
|
... )
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
# Generate deterministic UUID from memory_id (or random if not provided)
|
||||||
|
# Qdrant requires UUID or integer IDs, not arbitrary strings
|
||||||
|
if memory_id:
|
||||||
|
# Deterministic UUID from string - same memory_id = same UUID
|
||||||
|
point_id = str(uuid5(NAMESPACE_DNS, f"{user}:{memory_id}"))
|
||||||
|
else:
|
||||||
|
point_id = str(uuid4())
|
||||||
|
memory_id = point_id # Use UUID as the memory_id too
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Ensure collection exists
|
||||||
|
await self.ensure_collection(user)
|
||||||
|
|
||||||
|
# Create point (store original memory_id in payload for reference)
|
||||||
|
payload["memory_id"] = memory_id
|
||||||
|
point = qdrant_models.PointStruct(
|
||||||
|
id=point_id,
|
||||||
|
vector=vector,
|
||||||
|
payload=payload,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Upsert
|
||||||
|
self._client.upsert(
|
||||||
|
collection_name=collection_name,
|
||||||
|
points=[point],
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"qdrant_memory_upserted",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_id=memory_id,
|
||||||
|
memory_type=payload.get("type"),
|
||||||
|
)
|
||||||
|
return memory_id
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_upsert_memory_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_id=memory_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def search_memories(
|
||||||
|
self,
|
||||||
|
user: str,
|
||||||
|
query_vector: list[float],
|
||||||
|
limit: int = 10,
|
||||||
|
memory_type: str | None = None,
|
||||||
|
score_threshold: float = 0.5,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""
|
||||||
|
Search memories by vector similarity.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
query_vector: Query embedding vector
|
||||||
|
limit: Maximum results
|
||||||
|
memory_type: Filter by memory type (e.g., "fact", "preference", "profile")
|
||||||
|
score_threshold: Minimum similarity score (0-1)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of matching memories with scores
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> memories = await client.search_memories(
|
||||||
|
... user="jpmschweitzer",
|
||||||
|
... query_vector=[0.1, 0.2, ...],
|
||||||
|
... limit=5,
|
||||||
|
... memory_type="fact"
|
||||||
|
... )
|
||||||
|
>>> memories[0]
|
||||||
|
{"id": "mem_123", "score": 0.89, "type": "fact", "content": "..."}
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Build filter if memory_type specified
|
||||||
|
query_filter = None
|
||||||
|
if memory_type:
|
||||||
|
query_filter = qdrant_models.Filter(
|
||||||
|
must=[
|
||||||
|
qdrant_models.FieldCondition(
|
||||||
|
key="type",
|
||||||
|
match=qdrant_models.MatchValue(value=memory_type),
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
# Search using new Query API (qdrant-client >= 1.10)
|
||||||
|
results = self._client.query_points(
|
||||||
|
collection_name=collection_name,
|
||||||
|
query=query_vector,
|
||||||
|
limit=limit,
|
||||||
|
query_filter=query_filter,
|
||||||
|
score_threshold=score_threshold,
|
||||||
|
).points
|
||||||
|
|
||||||
|
# Format results
|
||||||
|
memories = []
|
||||||
|
for hit in results:
|
||||||
|
memory = {
|
||||||
|
"id": hit.id,
|
||||||
|
"score": hit.score,
|
||||||
|
**hit.payload,
|
||||||
|
}
|
||||||
|
memories.append(memory)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"qdrant_search_memories",
|
||||||
|
collection=collection_name,
|
||||||
|
results_count=len(memories),
|
||||||
|
memory_type=memory_type,
|
||||||
|
)
|
||||||
|
return memories
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_search_memories_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
async def get_memory(self, user: str, memory_id: str) -> dict[str, Any] | None:
|
||||||
|
"""
|
||||||
|
Get a specific memory by ID.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
memory_id: Memory ID
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Memory data or None if not found
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
# Convert memory_id to UUID point_id
|
||||||
|
point_id = str(uuid5(NAMESPACE_DNS, f"{user}:{memory_id}"))
|
||||||
|
|
||||||
|
try:
|
||||||
|
points = self._client.retrieve(
|
||||||
|
collection_name=collection_name,
|
||||||
|
ids=[point_id],
|
||||||
|
)
|
||||||
|
|
||||||
|
if not points:
|
||||||
|
return None
|
||||||
|
|
||||||
|
point = points[0]
|
||||||
|
return {
|
||||||
|
"id": point.id,
|
||||||
|
**point.payload,
|
||||||
|
}
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_get_memory_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_id=memory_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def delete_memory(self, user: str, memory_id: str) -> bool:
|
||||||
|
"""
|
||||||
|
Delete a memory by ID.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
memory_id: Memory ID to delete
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if deleted successfully, False otherwise
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> await client.delete_memory("jpmschweitzer", "mem_123")
|
||||||
|
True
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
# Convert memory_id to UUID point_id
|
||||||
|
point_id = str(uuid5(NAMESPACE_DNS, f"{user}:{memory_id}"))
|
||||||
|
|
||||||
|
try:
|
||||||
|
self._client.delete(
|
||||||
|
collection_name=collection_name,
|
||||||
|
points_selector=qdrant_models.PointIdsList(
|
||||||
|
points=[point_id],
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"qdrant_memory_deleted",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_id=memory_id,
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_delete_memory_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_id=memory_id,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def delete_memories_by_type(self, user: str, memory_type: str) -> int:
|
||||||
|
"""
|
||||||
|
Delete all memories of a specific type.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
memory_type: Type of memories to delete
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Number of memories deleted (approximate)
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Delete by filter
|
||||||
|
self._client.delete(
|
||||||
|
collection_name=collection_name,
|
||||||
|
points_selector=qdrant_models.FilterSelector(
|
||||||
|
filter=qdrant_models.Filter(
|
||||||
|
must=[
|
||||||
|
qdrant_models.FieldCondition(
|
||||||
|
key="type",
|
||||||
|
match=qdrant_models.MatchValue(value=memory_type),
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"qdrant_memories_deleted_by_type",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_type=memory_type,
|
||||||
|
)
|
||||||
|
return -1 # Qdrant doesn't return count for filter deletes
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_delete_memories_by_type_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
memory_type=memory_type,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
async def count_memories(self, user: str) -> int:
|
||||||
|
"""
|
||||||
|
Count total memories for a user.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
user: User identifier
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Number of memories in user's collection
|
||||||
|
"""
|
||||||
|
collection_name = get_memory_collection_name(user)
|
||||||
|
|
||||||
|
try:
|
||||||
|
info = self._client.get_collection(collection_name)
|
||||||
|
return info.points_count
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
"qdrant_count_memories_failed",
|
||||||
|
collection=collection_name,
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
async def health_check(self) -> bool:
|
||||||
|
"""
|
||||||
|
Check if Qdrant server is reachable.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if healthy, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
self._client.get_collections()
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("qdrant_health_check_failed", error=str(e))
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# Global client instance (lazy initialization)
|
||||||
|
_qdrant_client: MemoryQdrantClient | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def get_qdrant_client() -> MemoryQdrantClient:
|
||||||
|
"""
|
||||||
|
Get global Qdrant client instance.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
MemoryQdrantClient instance
|
||||||
|
"""
|
||||||
|
global _qdrant_client
|
||||||
|
if _qdrant_client is None:
|
||||||
|
_qdrant_client = MemoryQdrantClient()
|
||||||
|
return _qdrant_client
|
||||||
+3
-2
@@ -1,6 +1,7 @@
|
|||||||
"""
|
"""
|
||||||
Core router for health and root endpoints.
|
Core router for health and root endpoints.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
from fastapi import APIRouter
|
from fastapi import APIRouter
|
||||||
@@ -16,7 +17,7 @@ router = APIRouter(tags=["core"])
|
|||||||
async def health_check() -> dict[str, str]:
|
async def health_check() -> dict[str, str]:
|
||||||
"""
|
"""
|
||||||
Health check endpoint.
|
Health check endpoint.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Health status
|
Health status
|
||||||
"""
|
"""
|
||||||
@@ -27,7 +28,7 @@ async def health_check() -> dict[str, str]:
|
|||||||
async def root() -> dict[str, str]:
|
async def root() -> dict[str, str]:
|
||||||
"""
|
"""
|
||||||
Root endpoint.
|
Root endpoint.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
API information
|
API information
|
||||||
"""
|
"""
|
||||||
|
|||||||
+87
-10
@@ -5,14 +5,49 @@ Handles initialization of household registry and other startup tasks.
|
|||||||
This module should be called during application startup to register
|
This module should be called during application startup to register
|
||||||
all household members.
|
all household members.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from src.agents.biographer import register_biographer
|
||||||
|
from src.agents.housekeeper import register_housekeeper
|
||||||
|
from src.agents.librarian import register_librarian
|
||||||
from src.agents.tatlock_core import TATLOCK_CORE_CAPABILITY, tatlock_core_tools
|
from src.agents.tatlock_core import TATLOCK_CORE_CAPABILITY, tatlock_core_tools
|
||||||
|
from src.anthropic.model_selector import (
|
||||||
|
check_claude_health,
|
||||||
|
check_ollama_health,
|
||||||
|
get_model_info,
|
||||||
|
)
|
||||||
|
from src.core.config import Environment, config
|
||||||
from src.core.household_registry import get_household_registry
|
from src.core.household_registry import get_household_registry
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def register_household_members():
|
def log_tenant_guard() -> None:
|
||||||
|
"""
|
||||||
|
Emit one loud startup log line stating the effective tenant.
|
||||||
|
|
||||||
|
In non-production environments the tenant guard forces the reserved
|
||||||
|
test tenant regardless of DEFAULT_USER misconfiguration - this line
|
||||||
|
makes that override visible at startup.
|
||||||
|
"""
|
||||||
|
if config.ENVIRONMENT == Environment.PRODUCTION:
|
||||||
|
logger.info(
|
||||||
|
"tenant_guard_production",
|
||||||
|
environment=config.ENVIRONMENT.value,
|
||||||
|
tenant=config.effective_default_user,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
logger.warning(
|
||||||
|
"tenant_guard_active",
|
||||||
|
environment=config.ENVIRONMENT.value,
|
||||||
|
forced_tenant=config.effective_default_user,
|
||||||
|
default_user_overridden=config.tenant_forced,
|
||||||
|
configured_default_user=config.DEFAULT_USER,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def register_household_members() -> None:
|
||||||
"""
|
"""
|
||||||
Register all household members with the registry.
|
Register all household members with the registry.
|
||||||
|
|
||||||
@@ -21,11 +56,8 @@ def register_household_members():
|
|||||||
|
|
||||||
Currently registers:
|
Currently registers:
|
||||||
- tatlock_core: Butler's core tools (calculator, datetime, web search)
|
- tatlock_core: Butler's core tools (calculator, datetime, web search)
|
||||||
|
- librarian: Research and knowledge management (Phase 3)
|
||||||
Future phases will add:
|
- biographer: User memory and context management (Phase F)
|
||||||
- librarian: Research and knowledge management
|
|
||||||
- developer: Software development assistance
|
|
||||||
- etc.
|
|
||||||
"""
|
"""
|
||||||
registry = get_household_registry()
|
registry = get_household_registry()
|
||||||
|
|
||||||
@@ -45,25 +77,70 @@ def register_household_members():
|
|||||||
tool_count=len(tatlock_core_tools),
|
tool_count=len(tatlock_core_tools),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Register The Librarian (Phase 3)
|
||||||
|
try:
|
||||||
|
register_librarian()
|
||||||
|
except Exception as e:
|
||||||
|
# Don't fail startup if Librarian registration fails
|
||||||
|
logger.warning(
|
||||||
|
"librarian_registration_failed",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Register The Biographer (Phase F)
|
||||||
|
try:
|
||||||
|
register_biographer()
|
||||||
|
except Exception as e:
|
||||||
|
# Don't fail startup if Biographer registration fails
|
||||||
|
logger.warning(
|
||||||
|
"biographer_registration_failed",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Register The Housekeeper (Home Automation)
|
||||||
|
try:
|
||||||
|
register_housekeeper()
|
||||||
|
except Exception as e:
|
||||||
|
# Don't fail startup if Housekeeper registration fails
|
||||||
|
logger.warning(
|
||||||
|
"housekeeper_registration_failed",
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
"household_registration_complete",
|
"household_registration_complete",
|
||||||
total_members=len(registry),
|
total_members=len(registry),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def initialize_application():
|
async def initialize_application() -> None:
|
||||||
"""
|
"""
|
||||||
Initialize the application.
|
Initialize the application.
|
||||||
|
|
||||||
Performs all startup tasks:
|
Performs all startup tasks:
|
||||||
1. Register household members
|
1. Check Ollama (primary) and Claude (fallback) health for backend selection
|
||||||
2. (Future) Initialize connections
|
2. Register household members
|
||||||
3. (Future) Load configuration
|
3. (Future) Initialize connections
|
||||||
|
|
||||||
This should be called once during application startup.
|
This should be called once during application startup.
|
||||||
"""
|
"""
|
||||||
logger.info("application_initialization_starting")
|
logger.info("application_initialization_starting")
|
||||||
|
|
||||||
|
# Tenant isolation guard: state the effective tenant loudly
|
||||||
|
log_tenant_guard()
|
||||||
|
|
||||||
|
# Check backend health: Ollama is primary, Claude is the fallback
|
||||||
|
await check_ollama_health()
|
||||||
|
await check_claude_health()
|
||||||
|
model_info = get_model_info()
|
||||||
|
logger.info(
|
||||||
|
"model_backend_configured",
|
||||||
|
backend=model_info["backend"],
|
||||||
|
model=model_info["model"],
|
||||||
|
ollama_available=model_info["ollama_available"],
|
||||||
|
claude_available=model_info["claude_available"],
|
||||||
|
)
|
||||||
|
|
||||||
# Register household members
|
# Register household members
|
||||||
register_household_members()
|
register_household_members()
|
||||||
|
|
||||||
|
|||||||
+31
-58
@@ -1,13 +1,10 @@
|
|||||||
"""
|
"""
|
||||||
Tool call tracking and benchmarking.
|
Tool call tracking.
|
||||||
|
|
||||||
Tracks which tools are recommended by the Steward versus which tools
|
Tracks which tools are recommended by the Steward versus which tools
|
||||||
are actually used by Tatlock, recording benchmarks for analysis.
|
are actually used by Tatlock for debugging and analysis.
|
||||||
"""
|
"""
|
||||||
from datetime import datetime, timezone
|
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from src.core.benchmarks import PerformanceBenchmark, get_benchmark_store
|
|
||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
logger = get_logger(__name__)
|
logger = get_logger(__name__)
|
||||||
@@ -15,17 +12,13 @@ logger = get_logger(__name__)
|
|||||||
|
|
||||||
class ToolCallTracker:
|
class ToolCallTracker:
|
||||||
"""
|
"""
|
||||||
Tracks tool calls for benchmarking and accuracy analysis.
|
Tracks tool calls for accuracy analysis.
|
||||||
|
|
||||||
Compares Steward's recommendations with Tatlock's actual tool usage
|
Compares Steward's recommendations with Tatlock's actual tool usage
|
||||||
to measure recommendation accuracy.
|
to measure recommendation accuracy.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(
|
def __init__(self, recommended_capabilities: list[str], conversation_id: str | None = None):
|
||||||
self,
|
|
||||||
recommended_capabilities: list[str],
|
|
||||||
conversation_id: Optional[str] = None
|
|
||||||
):
|
|
||||||
"""
|
"""
|
||||||
Initialize tool call tracker.
|
Initialize tool call tracker.
|
||||||
|
|
||||||
@@ -43,7 +36,21 @@ class ToolCallTracker:
|
|||||||
conversation_id=conversation_id,
|
conversation_id=conversation_id,
|
||||||
)
|
)
|
||||||
|
|
||||||
async def track_call(self, tool_name: str, duration: float):
|
def _extract_capability(self, tool_name: str) -> str:
|
||||||
|
"""
|
||||||
|
Extract capability name from tool name.
|
||||||
|
|
||||||
|
Tool names like 'delegate_to_librarian' map to capability 'librarian'.
|
||||||
|
"""
|
||||||
|
if tool_name.startswith("delegate_to_"):
|
||||||
|
return tool_name.replace("delegate_to_", "")
|
||||||
|
return tool_name
|
||||||
|
|
||||||
|
def log_call(self, message: str) -> None:
|
||||||
|
"""Log a tool call message (for UI display)."""
|
||||||
|
logger.debug("tool_call_message", message=message)
|
||||||
|
|
||||||
|
async def track_call(self, tool_name: str, duration: float) -> None:
|
||||||
"""
|
"""
|
||||||
Record a tool call with timing.
|
Record a tool call with timing.
|
||||||
|
|
||||||
@@ -56,8 +63,9 @@ class ToolCallTracker:
|
|||||||
self.actual_calls[tool_name] = []
|
self.actual_calls[tool_name] = []
|
||||||
self.actual_calls[tool_name].append(duration)
|
self.actual_calls[tool_name].append(duration)
|
||||||
|
|
||||||
# Check if tool was recommended
|
# Check if tool was recommended (normalize tool name to capability)
|
||||||
was_recommended = tool_name in self.recommended_capabilities
|
capability = self._extract_capability(tool_name)
|
||||||
|
was_recommended = capability in self.recommended_capabilities
|
||||||
|
|
||||||
if not was_recommended:
|
if not was_recommended:
|
||||||
logger.warning(
|
logger.warning(
|
||||||
@@ -67,23 +75,6 @@ class ToolCallTracker:
|
|||||||
recommended=list(self.recommended_capabilities),
|
recommended=list(self.recommended_capabilities),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Record benchmark to Redis
|
|
||||||
benchmark = PerformanceBenchmark(
|
|
||||||
timestamp=datetime.now(timezone.utc),
|
|
||||||
operation="tool_call",
|
|
||||||
duration_seconds=duration,
|
|
||||||
success=True, # If we got here, the call succeeded
|
|
||||||
tool_name=tool_name,
|
|
||||||
was_recommended=was_recommended,
|
|
||||||
was_actually_used=True,
|
|
||||||
conversation_id=self.conversation_id,
|
|
||||||
metadata={
|
|
||||||
"recommended_capabilities": list(self.recommended_capabilities),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
|
|
||||||
await get_benchmark_store().record(benchmark)
|
|
||||||
|
|
||||||
logger.debug(
|
logger.debug(
|
||||||
"tool_call_tracked",
|
"tool_call_tracked",
|
||||||
tool_name=tool_name,
|
tool_name=tool_name,
|
||||||
@@ -91,15 +82,17 @@ class ToolCallTracker:
|
|||||||
was_recommended=was_recommended,
|
was_recommended=was_recommended,
|
||||||
)
|
)
|
||||||
|
|
||||||
async def finalize(self):
|
async def finalize(self) -> None:
|
||||||
"""
|
"""
|
||||||
Finalize tracking and log unused recommended tools.
|
Finalize tracking and log unused recommended tools.
|
||||||
|
|
||||||
Called after Tatlock completes its response to identify
|
Called after Tatlock completes its response to identify
|
||||||
tools that were recommended but never used.
|
tools that were recommended but never used.
|
||||||
"""
|
"""
|
||||||
|
# Normalize actual tool names to capabilities for comparison
|
||||||
|
used_capabilities = {self._extract_capability(tool) for tool in self.actual_calls.keys()}
|
||||||
# Find tools that were recommended but not used
|
# Find tools that were recommended but not used
|
||||||
unused_tools = self.recommended_capabilities - set(self.actual_calls.keys())
|
unused_tools = self.recommended_capabilities - used_capabilities
|
||||||
|
|
||||||
if unused_tools:
|
if unused_tools:
|
||||||
logger.info(
|
logger.info(
|
||||||
@@ -109,24 +102,6 @@ class ToolCallTracker:
|
|||||||
conversation_id=self.conversation_id,
|
conversation_id=self.conversation_id,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Record benchmarks for unused recommendations
|
|
||||||
for tool_name in unused_tools:
|
|
||||||
benchmark = PerformanceBenchmark(
|
|
||||||
timestamp=datetime.now(timezone.utc),
|
|
||||||
operation="tool_call",
|
|
||||||
duration_seconds=0.0, # Not used
|
|
||||||
success=True,
|
|
||||||
tool_name=tool_name,
|
|
||||||
was_recommended=True,
|
|
||||||
was_actually_used=False,
|
|
||||||
conversation_id=self.conversation_id,
|
|
||||||
metadata={
|
|
||||||
"recommended_capabilities": list(self.recommended_capabilities),
|
|
||||||
"reason": "recommended_but_unused",
|
|
||||||
},
|
|
||||||
)
|
|
||||||
await get_benchmark_store().record(benchmark)
|
|
||||||
|
|
||||||
# Log summary
|
# Log summary
|
||||||
total_calls = sum(len(durations) for durations in self.actual_calls.values())
|
total_calls = sum(len(durations) for durations in self.actual_calls.values())
|
||||||
logger.info(
|
logger.info(
|
||||||
@@ -145,7 +120,9 @@ class ToolCallTracker:
|
|||||||
Dict with tracking statistics
|
Dict with tracking statistics
|
||||||
"""
|
"""
|
||||||
total_calls = sum(len(durations) for durations in self.actual_calls.values())
|
total_calls = sum(len(durations) for durations in self.actual_calls.values())
|
||||||
unused = self.recommended_capabilities - set(self.actual_calls.keys())
|
# Normalize actual tool names to capabilities for comparison
|
||||||
|
used_capabilities = {self._extract_capability(tool) for tool in self.actual_calls.keys()}
|
||||||
|
unused = self.recommended_capabilities - used_capabilities
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"recommended_capabilities": list(self.recommended_capabilities),
|
"recommended_capabilities": list(self.recommended_capabilities),
|
||||||
@@ -153,12 +130,8 @@ class ToolCallTracker:
|
|||||||
"tools_unused": list(unused),
|
"tools_unused": list(unused),
|
||||||
"total_calls": total_calls,
|
"total_calls": total_calls,
|
||||||
"accuracy": {
|
"accuracy": {
|
||||||
"recommended_and_used": len(
|
"recommended_and_used": len(self.recommended_capabilities & used_capabilities),
|
||||||
self.recommended_capabilities & set(self.actual_calls.keys())
|
|
||||||
),
|
|
||||||
"recommended_but_unused": len(unused),
|
"recommended_but_unused": len(unused),
|
||||||
"not_recommended_but_used": len(
|
"not_recommended_but_used": len(used_capabilities - self.recommended_capabilities),
|
||||||
set(self.actual_calls.keys()) - self.recommended_capabilities
|
|
||||||
),
|
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,443 @@
|
|||||||
|
"""
|
||||||
|
Lightweight request tracing for local development.
|
||||||
|
|
||||||
|
Captures the full request flow through Tatlock's multi-agent architecture
|
||||||
|
as structured JSON traces for debugging and optimization.
|
||||||
|
|
||||||
|
Enable via DEBUG=true environment variable.
|
||||||
|
|
||||||
|
Traces are written to logs/traces/{trace_id}.json
|
||||||
|
View with logs/traces/viewer.html
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import secrets
|
||||||
|
from contextlib import asynccontextmanager
|
||||||
|
from contextvars import ContextVar
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from enum import Enum
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class SpanType(str, Enum):
|
||||||
|
"""Types of traced operations."""
|
||||||
|
|
||||||
|
ROUTER = "router"
|
||||||
|
STEWARD = "steward"
|
||||||
|
TATLOCK = "tatlock"
|
||||||
|
EXPERT = "expert"
|
||||||
|
TOOL = "tool"
|
||||||
|
|
||||||
|
|
||||||
|
class SpanStatus(str, Enum):
|
||||||
|
"""Span completion status."""
|
||||||
|
|
||||||
|
OK = "ok"
|
||||||
|
ERROR = "error"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Span:
|
||||||
|
"""A single traced operation."""
|
||||||
|
|
||||||
|
span_id: str
|
||||||
|
name: str
|
||||||
|
type: SpanType
|
||||||
|
start_time: datetime
|
||||||
|
parent_id: str | None = None
|
||||||
|
end_time: datetime | None = None
|
||||||
|
status: SpanStatus = SpanStatus.OK
|
||||||
|
metadata: dict[str, Any] = field(default_factory=dict)
|
||||||
|
details: dict[str, Any] = field(default_factory=dict)
|
||||||
|
children: list[str] = field(default_factory=list)
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def duration_ms(self) -> float | None:
|
||||||
|
"""Calculate duration in milliseconds."""
|
||||||
|
if self.end_time and self.start_time:
|
||||||
|
return (self.end_time - self.start_time).total_seconds() * 1000
|
||||||
|
return None
|
||||||
|
|
||||||
|
def to_dict(self) -> dict[str, Any]:
|
||||||
|
"""Convert span to dictionary for JSON serialization."""
|
||||||
|
result = {
|
||||||
|
"span_id": self.span_id,
|
||||||
|
"parent_id": self.parent_id,
|
||||||
|
"name": self.name,
|
||||||
|
"type": self.type.value,
|
||||||
|
"start_time": self.start_time.isoformat(),
|
||||||
|
"end_time": self.end_time.isoformat() if self.end_time else None,
|
||||||
|
"duration_ms": round(self.duration_ms, 2) if self.duration_ms else None,
|
||||||
|
"status": self.status.value,
|
||||||
|
"metadata": self.metadata if self.metadata else None,
|
||||||
|
}
|
||||||
|
# Only include non-empty optional fields
|
||||||
|
if self.details:
|
||||||
|
result["details"] = self.details
|
||||||
|
if self.children:
|
||||||
|
result["children"] = self.children
|
||||||
|
if self.error:
|
||||||
|
result["error"] = self.error
|
||||||
|
return {k: v for k, v in result.items() if v is not None}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Trace:
|
||||||
|
"""Complete trace of a request."""
|
||||||
|
|
||||||
|
trace_id: str
|
||||||
|
conversation_id: str | None
|
||||||
|
user: str
|
||||||
|
timestamp: datetime
|
||||||
|
request: dict[str, Any]
|
||||||
|
spans: list[Span] = field(default_factory=list)
|
||||||
|
response: dict[str, Any] | None = None
|
||||||
|
status: str = "in_progress"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def total_duration_ms(self) -> float | None:
|
||||||
|
"""Calculate total trace duration from span timings."""
|
||||||
|
if not self.spans:
|
||||||
|
return None
|
||||||
|
start = min(s.start_time for s in self.spans)
|
||||||
|
ends = [s.end_time for s in self.spans if s.end_time]
|
||||||
|
if not ends:
|
||||||
|
return None
|
||||||
|
end = max(ends)
|
||||||
|
return (end - start).total_seconds() * 1000
|
||||||
|
|
||||||
|
def to_dict(self) -> dict[str, Any]:
|
||||||
|
"""Convert trace to dictionary for JSON serialization."""
|
||||||
|
return {
|
||||||
|
"trace_id": self.trace_id,
|
||||||
|
"conversation_id": self.conversation_id,
|
||||||
|
"user": self.user,
|
||||||
|
"timestamp": self.timestamp.isoformat(),
|
||||||
|
"total_duration_ms": round(self.total_duration_ms, 2)
|
||||||
|
if self.total_duration_ms
|
||||||
|
else None,
|
||||||
|
"status": self.status,
|
||||||
|
"request": self.request,
|
||||||
|
"response": self.response,
|
||||||
|
"spans": [s.to_dict() for s in self.spans],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ContextVar for async-safe trace propagation
|
||||||
|
_current_trace: ContextVar[Trace | None] = ContextVar("current_trace", default=None)
|
||||||
|
_current_span: ContextVar[Span | None] = ContextVar("current_span", default=None)
|
||||||
|
|
||||||
|
|
||||||
|
def tracing_enabled() -> bool:
|
||||||
|
"""Check if tracing is enabled (requires DEBUG=true)."""
|
||||||
|
from src.core.config import config
|
||||||
|
|
||||||
|
return config.DEBUG
|
||||||
|
|
||||||
|
|
||||||
|
def _generate_id(prefix: str = "") -> str:
|
||||||
|
"""Generate unique ID with optional prefix."""
|
||||||
|
return f"{prefix}{secrets.token_hex(8)}"
|
||||||
|
|
||||||
|
|
||||||
|
def start_trace(
|
||||||
|
conversation_id: str | None,
|
||||||
|
user: str,
|
||||||
|
request: dict[str, Any],
|
||||||
|
) -> Trace | None:
|
||||||
|
"""
|
||||||
|
Start a new trace for a request.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
conversation_id: Conversation identifier
|
||||||
|
user: User identifier
|
||||||
|
request: Request data (should include preview and full)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Trace object if tracing enabled, None otherwise
|
||||||
|
"""
|
||||||
|
if not tracing_enabled():
|
||||||
|
return None
|
||||||
|
|
||||||
|
trace = Trace(
|
||||||
|
trace_id=_generate_id("trace_"),
|
||||||
|
conversation_id=conversation_id,
|
||||||
|
user=user,
|
||||||
|
timestamp=datetime.now(UTC),
|
||||||
|
request=request,
|
||||||
|
)
|
||||||
|
_current_trace.set(trace)
|
||||||
|
|
||||||
|
logger.debug("trace_started", trace_id=trace.trace_id, user=user)
|
||||||
|
return trace
|
||||||
|
|
||||||
|
|
||||||
|
def get_current_trace() -> Trace | None:
|
||||||
|
"""Get the current trace from context."""
|
||||||
|
return _current_trace.get()
|
||||||
|
|
||||||
|
|
||||||
|
def get_current_span() -> Span | None:
|
||||||
|
"""Get the current span from context."""
|
||||||
|
return _current_span.get()
|
||||||
|
|
||||||
|
|
||||||
|
def start_span(
|
||||||
|
name: str,
|
||||||
|
span_type: SpanType,
|
||||||
|
metadata: dict[str, Any] | None = None,
|
||||||
|
details: dict[str, Any] | None = None,
|
||||||
|
) -> Span | None:
|
||||||
|
"""
|
||||||
|
Start a new span within the current trace.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
name: Span name (e.g., "steward_analysis")
|
||||||
|
span_type: Type of operation
|
||||||
|
metadata: Quick-access metadata (shown in timeline)
|
||||||
|
details: Expandable details (prompts, full responses)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Span object if tracing enabled, None otherwise
|
||||||
|
"""
|
||||||
|
trace = get_current_trace()
|
||||||
|
if not trace:
|
||||||
|
return None
|
||||||
|
|
||||||
|
parent = get_current_span()
|
||||||
|
span = Span(
|
||||||
|
span_id=_generate_id("span_"),
|
||||||
|
name=name,
|
||||||
|
type=span_type,
|
||||||
|
start_time=datetime.now(UTC),
|
||||||
|
parent_id=parent.span_id if parent else None,
|
||||||
|
metadata=metadata or {},
|
||||||
|
details=details or {},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Add to parent's children list
|
||||||
|
if parent:
|
||||||
|
parent.children.append(span.span_id)
|
||||||
|
|
||||||
|
trace.spans.append(span)
|
||||||
|
_current_span.set(span)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"span_started",
|
||||||
|
span_id=span.span_id,
|
||||||
|
name=name,
|
||||||
|
type=span_type.value,
|
||||||
|
parent_id=span.parent_id,
|
||||||
|
)
|
||||||
|
return span
|
||||||
|
|
||||||
|
|
||||||
|
def end_span(
|
||||||
|
span: Span | None = None,
|
||||||
|
status: SpanStatus = SpanStatus.OK,
|
||||||
|
metadata_update: dict[str, Any] | None = None,
|
||||||
|
details_update: dict[str, Any] | None = None,
|
||||||
|
error: str | None = None,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
End a span and restore parent as current.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
span: Span to end (defaults to current span)
|
||||||
|
status: Completion status
|
||||||
|
metadata_update: Additional metadata to merge
|
||||||
|
details_update: Additional details to merge
|
||||||
|
error: Error message if failed
|
||||||
|
"""
|
||||||
|
if span is None:
|
||||||
|
span = get_current_span()
|
||||||
|
if not span:
|
||||||
|
return
|
||||||
|
|
||||||
|
span.end_time = datetime.now(UTC)
|
||||||
|
span.status = status
|
||||||
|
if error:
|
||||||
|
span.error = error
|
||||||
|
span.status = SpanStatus.ERROR
|
||||||
|
if metadata_update:
|
||||||
|
span.metadata.update(metadata_update)
|
||||||
|
if details_update:
|
||||||
|
span.details.update(details_update)
|
||||||
|
|
||||||
|
# Restore parent span as current
|
||||||
|
trace = get_current_trace()
|
||||||
|
if trace and span.parent_id:
|
||||||
|
parent = next((s for s in trace.spans if s.span_id == span.parent_id), None)
|
||||||
|
_current_span.set(parent)
|
||||||
|
else:
|
||||||
|
_current_span.set(None)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
"span_ended",
|
||||||
|
span_id=span.span_id,
|
||||||
|
duration_ms=span.duration_ms,
|
||||||
|
status=status.value,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def end_trace(
|
||||||
|
response: dict[str, Any] | None = None,
|
||||||
|
status: str = "completed",
|
||||||
|
) -> str | None:
|
||||||
|
"""
|
||||||
|
End the current trace and write to file.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
response: Response data to include
|
||||||
|
status: Final trace status ("completed" or "error")
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path to trace file if written, None otherwise
|
||||||
|
"""
|
||||||
|
trace = get_current_trace()
|
||||||
|
if not trace:
|
||||||
|
return None
|
||||||
|
|
||||||
|
trace.response = response
|
||||||
|
trace.status = status
|
||||||
|
|
||||||
|
# Write trace to file
|
||||||
|
trace_path = _write_trace(trace)
|
||||||
|
|
||||||
|
# Clear context
|
||||||
|
_current_trace.set(None)
|
||||||
|
_current_span.set(None)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"trace_completed",
|
||||||
|
trace_id=trace.trace_id,
|
||||||
|
total_duration_ms=round(trace.total_duration_ms, 2) if trace.total_duration_ms else None,
|
||||||
|
span_count=len(trace.spans),
|
||||||
|
path=str(trace_path) if trace_path else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
return str(trace_path) if trace_path else None
|
||||||
|
|
||||||
|
|
||||||
|
def _write_trace(trace: Trace) -> Path | None:
|
||||||
|
"""Write trace to JSON file."""
|
||||||
|
try:
|
||||||
|
# Ensure traces directory exists
|
||||||
|
traces_dir = Path("logs/traces")
|
||||||
|
traces_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
# Write trace file
|
||||||
|
trace_path = traces_dir / f"{trace.trace_id}.json"
|
||||||
|
with open(trace_path, "w") as f:
|
||||||
|
json.dump(trace.to_dict(), f, indent=2, default=str)
|
||||||
|
|
||||||
|
return trace_path
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("trace_write_failed", error=str(e), trace_id=trace.trace_id)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
@asynccontextmanager
|
||||||
|
async def trace_span(
|
||||||
|
name: str,
|
||||||
|
span_type: SpanType,
|
||||||
|
metadata: dict[str, Any] | None = None,
|
||||||
|
details: dict[str, Any] | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Async context manager for tracing a span.
|
||||||
|
|
||||||
|
Automatically handles start/end timing and error capture.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
async with trace_span("steward_analysis", SpanType.STEWARD) as span:
|
||||||
|
result = await analyze_request(...)
|
||||||
|
if span:
|
||||||
|
span.metadata["result_count"] = len(result)
|
||||||
|
|
||||||
|
Args:
|
||||||
|
name: Span name
|
||||||
|
span_type: Type of operation
|
||||||
|
metadata: Initial metadata
|
||||||
|
details: Initial details (expandable in viewer)
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
Span object or None if tracing disabled
|
||||||
|
"""
|
||||||
|
span = start_span(name, span_type, metadata, details)
|
||||||
|
try:
|
||||||
|
yield span
|
||||||
|
except Exception as e:
|
||||||
|
end_span(span, SpanStatus.ERROR, error=str(e))
|
||||||
|
raise
|
||||||
|
else:
|
||||||
|
end_span(span, SpanStatus.OK)
|
||||||
|
|
||||||
|
|
||||||
|
def add_tool_spans_from_messages(messages: list[Any], parent_span: Span | None = None) -> None:
|
||||||
|
"""
|
||||||
|
Extract tool calls from PydanticAI result messages and add as child spans.
|
||||||
|
|
||||||
|
Call this after an agent.run() to capture tool-level timing retroactively.
|
||||||
|
Note: Since we don't have actual timing, we estimate based on sequence.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
messages: List from result.new_messages()
|
||||||
|
parent_span: Parent span to attach tool spans to
|
||||||
|
"""
|
||||||
|
trace = get_current_trace()
|
||||||
|
if not trace or not parent_span:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Import PydanticAI message types
|
||||||
|
try:
|
||||||
|
from pydantic_ai.messages import ModelRequest, ModelResponse, ToolCallPart, ToolReturnPart
|
||||||
|
except ImportError:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Track tool calls and their returns
|
||||||
|
tool_calls: dict[str, dict[str, Any]] = {}
|
||||||
|
|
||||||
|
for msg in messages:
|
||||||
|
if isinstance(msg, ModelResponse):
|
||||||
|
for part in msg.parts:
|
||||||
|
if isinstance(part, ToolCallPart):
|
||||||
|
tool_calls[part.tool_call_id] = {
|
||||||
|
"name": part.tool_name,
|
||||||
|
"args": part.args if hasattr(part, "args") else {},
|
||||||
|
}
|
||||||
|
elif isinstance(msg, ModelRequest):
|
||||||
|
for part in msg.parts:
|
||||||
|
if isinstance(part, ToolReturnPart):
|
||||||
|
if part.tool_call_id in tool_calls:
|
||||||
|
tool_info = tool_calls[part.tool_call_id]
|
||||||
|
# Create a span for this tool call
|
||||||
|
span = Span(
|
||||||
|
span_id=_generate_id("span_"),
|
||||||
|
name=tool_info["name"],
|
||||||
|
type=SpanType.TOOL,
|
||||||
|
start_time=parent_span.start_time, # Approximate
|
||||||
|
end_time=parent_span.end_time or datetime.now(UTC),
|
||||||
|
parent_id=parent_span.span_id,
|
||||||
|
status=SpanStatus.OK,
|
||||||
|
metadata={
|
||||||
|
"tool_name": tool_info["name"],
|
||||||
|
"args_preview": str(tool_info.get("args", {}))[:100],
|
||||||
|
},
|
||||||
|
details={
|
||||||
|
"args": tool_info.get("args", {}),
|
||||||
|
"result": part.content[:2000]
|
||||||
|
if isinstance(part.content, str)
|
||||||
|
else str(part.content)[:2000],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
parent_span.children.append(span.span_id)
|
||||||
|
trace.spans.append(span)
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
"""
|
||||||
|
Trace viewer router.
|
||||||
|
|
||||||
|
Serves the trace viewer UI and trace files when tracing is enabled.
|
||||||
|
Only available when DEBUG=true.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import UTC
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from fastapi import APIRouter, HTTPException
|
||||||
|
from fastapi.responses import HTMLResponse, JSONResponse
|
||||||
|
|
||||||
|
from src.core.config import config
|
||||||
|
from src.core.logging_config import get_logger
|
||||||
|
|
||||||
|
logger = get_logger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter(prefix="/traces", tags=["traces"])
|
||||||
|
|
||||||
|
TRACES_DIR = Path("logs/traces")
|
||||||
|
VIEWER_PATH = TRACES_DIR / "viewer.html"
|
||||||
|
|
||||||
|
|
||||||
|
def tracing_enabled() -> bool:
|
||||||
|
"""Check if tracing is enabled."""
|
||||||
|
return config.DEBUG
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("", response_class=HTMLResponse)
|
||||||
|
async def get_trace_viewer():
|
||||||
|
"""
|
||||||
|
Serve the trace viewer UI.
|
||||||
|
|
||||||
|
Returns the standalone HTML viewer for browsing traces.
|
||||||
|
"""
|
||||||
|
if not tracing_enabled():
|
||||||
|
raise HTTPException(status_code=404, detail="Tracing not enabled")
|
||||||
|
|
||||||
|
if not VIEWER_PATH.exists():
|
||||||
|
raise HTTPException(status_code=404, detail="Viewer not found")
|
||||||
|
|
||||||
|
return HTMLResponse(content=VIEWER_PATH.read_text())
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/list")
|
||||||
|
async def list_traces(
|
||||||
|
limit: int = 50,
|
||||||
|
since_minutes: int | None = None,
|
||||||
|
status: str | None = None,
|
||||||
|
search: str | None = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
List available trace files.
|
||||||
|
|
||||||
|
Returns most recent traces first, with basic metadata.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
limit: Maximum number of traces to return (default 50)
|
||||||
|
since_minutes: Only return traces from the last N minutes
|
||||||
|
status: Filter by status (completed, error, streaming)
|
||||||
|
search: Search in request preview text
|
||||||
|
"""
|
||||||
|
if not tracing_enabled():
|
||||||
|
raise HTTPException(status_code=404, detail="Tracing not enabled")
|
||||||
|
|
||||||
|
if not TRACES_DIR.exists():
|
||||||
|
return {"traces": [], "total": 0}
|
||||||
|
|
||||||
|
import json
|
||||||
|
from datetime import datetime, timedelta
|
||||||
|
|
||||||
|
# Calculate cutoff time if filtering by time
|
||||||
|
cutoff_time = None
|
||||||
|
if since_minutes:
|
||||||
|
cutoff_time = datetime.now(UTC) - timedelta(minutes=since_minutes)
|
||||||
|
|
||||||
|
# Get all trace files, sorted by modification time (newest first)
|
||||||
|
trace_files = sorted(
|
||||||
|
TRACES_DIR.glob("trace_*.json"),
|
||||||
|
key=lambda p: p.stat().st_mtime,
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
traces: list[dict[str, Any]] = []
|
||||||
|
for path in trace_files:
|
||||||
|
if len(traces) >= limit:
|
||||||
|
break
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
|
||||||
|
# Parse timestamp for filtering
|
||||||
|
trace_timestamp = data.get("timestamp")
|
||||||
|
if cutoff_time and trace_timestamp:
|
||||||
|
try:
|
||||||
|
ts = datetime.fromisoformat(trace_timestamp.replace("Z", "+00:00"))
|
||||||
|
if ts < cutoff_time:
|
||||||
|
continue
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Filter by status
|
||||||
|
trace_status = data.get("status", "")
|
||||||
|
if status and trace_status != status:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Filter by search text
|
||||||
|
request_preview = data.get("request", {}).get("input_preview", "")
|
||||||
|
if search and search.lower() not in request_preview.lower():
|
||||||
|
continue
|
||||||
|
|
||||||
|
traces.append(
|
||||||
|
{
|
||||||
|
"trace_id": data.get("trace_id"),
|
||||||
|
"timestamp": trace_timestamp,
|
||||||
|
"user": data.get("user"),
|
||||||
|
"status": trace_status,
|
||||||
|
"total_duration_ms": data.get("total_duration_ms"),
|
||||||
|
"span_count": len(data.get("spans", [])),
|
||||||
|
"request_preview": request_preview[:100],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("trace_list_parse_error", path=str(path), error=str(e))
|
||||||
|
|
||||||
|
return {"traces": traces, "total": len(traces)}
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/{trace_id}")
|
||||||
|
async def get_trace(trace_id: str):
|
||||||
|
"""
|
||||||
|
Get a specific trace by ID.
|
||||||
|
|
||||||
|
Returns the full trace JSON.
|
||||||
|
"""
|
||||||
|
if not tracing_enabled():
|
||||||
|
raise HTTPException(status_code=404, detail="Tracing not enabled")
|
||||||
|
|
||||||
|
# Sanitize trace_id to prevent path traversal
|
||||||
|
if not trace_id.startswith("trace_") or "/" in trace_id or "\\" in trace_id:
|
||||||
|
raise HTTPException(status_code=400, detail="Invalid trace ID")
|
||||||
|
|
||||||
|
trace_path = TRACES_DIR / f"{trace_id}.json"
|
||||||
|
|
||||||
|
if not trace_path.exists():
|
||||||
|
raise HTTPException(status_code=404, detail="Trace not found")
|
||||||
|
|
||||||
|
try:
|
||||||
|
import json
|
||||||
|
|
||||||
|
with open(trace_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return JSONResponse(content=data)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("trace_read_error", trace_id=trace_id, error=str(e))
|
||||||
|
raise HTTPException(status_code=500, detail="Failed to read trace") from e
|
||||||
+20
-11
@@ -9,8 +9,9 @@ Main responsibilities:
|
|||||||
- Router registration
|
- Router registration
|
||||||
- Lifecycle management
|
- Lifecycle management
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from collections.abc import AsyncGenerator
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
from typing import AsyncGenerator
|
|
||||||
|
|
||||||
from fastapi import FastAPI, Request, status
|
from fastapi import FastAPI, Request, status
|
||||||
from fastapi.exceptions import RequestValidationError
|
from fastapi.exceptions import RequestValidationError
|
||||||
@@ -23,6 +24,7 @@ from src.core.exceptions import AppException
|
|||||||
from src.core.logging_config import get_logger
|
from src.core.logging_config import get_logger
|
||||||
from src.core.router import router as core_router
|
from src.core.router import router as core_router
|
||||||
from src.core.startup import initialize_application
|
from src.core.startup import initialize_application
|
||||||
|
from src.core.tracing_router import router as tracing_router
|
||||||
from src.models.router import router as models_router
|
from src.models.router import router as models_router
|
||||||
from src.responses.router import router as responses_router
|
from src.responses.router import router as responses_router
|
||||||
|
|
||||||
@@ -43,14 +45,16 @@ async def lifespan(app: FastAPI) -> AsyncGenerator[None, None]:
|
|||||||
app_name=config.APP_NAME,
|
app_name=config.APP_NAME,
|
||||||
version=config.APP_VERSION,
|
version=config.APP_VERSION,
|
||||||
environment=config.ENVIRONMENT.value,
|
environment=config.ENVIRONMENT.value,
|
||||||
|
prefer_cloud=config.PREFER_CLOUD_BACKEND,
|
||||||
|
anthropic_model=config.ANTHROPIC_MODEL,
|
||||||
ollama_host=str(config.OLLAMA_HOST),
|
ollama_host=str(config.OLLAMA_HOST),
|
||||||
ollama_model=config.OLLAMA_DEFAULT_MODEL,
|
ollama_model=config.OLLAMA_DEFAULT_MODEL,
|
||||||
redis_url=config.redis_url,
|
redis_url=config.redis_memory_url,
|
||||||
log_format=config.log_format,
|
log_format=config.log_format,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Initialize application (register household members, etc.)
|
# Initialize application (check Claude health, register household members, etc.)
|
||||||
initialize_application()
|
await initialize_application()
|
||||||
|
|
||||||
yield
|
yield
|
||||||
|
|
||||||
@@ -61,7 +65,7 @@ async def lifespan(app: FastAPI) -> AsyncGenerator[None, None]:
|
|||||||
def create_application() -> FastAPI:
|
def create_application() -> FastAPI:
|
||||||
"""
|
"""
|
||||||
Application factory.
|
Application factory.
|
||||||
|
|
||||||
Creates and configures the FastAPI application.
|
Creates and configures the FastAPI application.
|
||||||
Following best practice of using factory pattern.
|
Following best practice of using factory pattern.
|
||||||
"""
|
"""
|
||||||
@@ -72,7 +76,7 @@ def create_application() -> FastAPI:
|
|||||||
lifespan=lifespan,
|
lifespan=lifespan,
|
||||||
debug=config.DEBUG,
|
debug=config.DEBUG,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Add middleware
|
# Add middleware
|
||||||
application.add_middleware(
|
application.add_middleware(
|
||||||
CORSMiddleware,
|
CORSMiddleware,
|
||||||
@@ -81,26 +85,31 @@ def create_application() -> FastAPI:
|
|||||||
allow_methods=config.CORS_ALLOW_METHODS,
|
allow_methods=config.CORS_ALLOW_METHODS,
|
||||||
allow_headers=config.CORS_ALLOW_HEADERS,
|
allow_headers=config.CORS_ALLOW_HEADERS,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Register exception handlers
|
# Register exception handlers
|
||||||
register_exception_handlers(application)
|
register_exception_handlers(application)
|
||||||
|
|
||||||
# Include routers
|
# Include routers
|
||||||
application.include_router(core_router) # Health and root endpoints
|
application.include_router(core_router) # Health and root endpoints
|
||||||
application.include_router(chat_router, prefix=config.API_PREFIX)
|
application.include_router(chat_router, prefix=config.API_PREFIX)
|
||||||
application.include_router(models_router, prefix=config.API_PREFIX)
|
application.include_router(models_router, prefix=config.API_PREFIX)
|
||||||
application.include_router(responses_router, prefix=config.API_PREFIX) # Responses API
|
application.include_router(responses_router, prefix=config.API_PREFIX) # Responses API
|
||||||
|
|
||||||
|
# Conditionally include tracing router (only in debug mode)
|
||||||
|
if config.DEBUG:
|
||||||
|
application.include_router(tracing_router)
|
||||||
|
logger.info("tracing_router_enabled")
|
||||||
|
|
||||||
return application
|
return application
|
||||||
|
|
||||||
|
|
||||||
def register_exception_handlers(application: FastAPI) -> None:
|
def register_exception_handlers(application: FastAPI) -> None:
|
||||||
"""
|
"""
|
||||||
Register global exception handlers.
|
Register global exception handlers.
|
||||||
|
|
||||||
Provides consistent error responses compatible with OpenAI API.
|
Provides consistent error responses compatible with OpenAI API.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@application.exception_handler(AppException)
|
@application.exception_handler(AppException)
|
||||||
async def app_exception_handler(
|
async def app_exception_handler(
|
||||||
request: Request,
|
request: Request,
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user