diff --git a/.env.example b/.env.example index 184054595..a7d1074a7 100644 --- a/.env.example +++ b/.env.example @@ -79,7 +79,7 @@ SEARXNG_INSTANCE=http://localhost:8080 # Keep APP_BIND on loopback unless you intentionally want LAN/reverse-proxy access. # APP_BIND=127.0.0.1 # Change this if another local service already uses 7000 (macOS AirPlay often does). -# APP_PORT=7000 +# APP_PORT=7011 # Optional HTTP address advertised in companion/mobile pairing codes. Set this # when Docker would otherwise advertise a container address or loopback. Use a @@ -95,7 +95,6 @@ SEARXNG_INSTANCE=http://localhost:8080 # Skip the external-context exact-approval pause for unattended local agents. # Keep false for shared or internet-exposed deployments. -# ODYSSEUS_UNATTENDED_MODE=false # Optional post-external-context tool approval gate. Off by default because it # can block normal agent work; enable only for deployments that want this fence. @@ -277,7 +276,7 @@ SEARXNG_INSTANCE=http://localhost:8080 # are reached through their host-published loopback ports. # COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml:docker/host-network.yml # APP_BIND=127.0.0.1 -# APP_PORT=7000 +# APP_PORT=7011 # ODYSSEUS_HOST_NETWORK_SEARXNG_INSTANCE=http://127.0.0.1:8080 # ODYSSEUS_HOST_NETWORK_CHROMADB_HOST=127.0.0.1 # ODYSSEUS_HOST_NETWORK_CHROMADB_PORT=8100 diff --git a/README.md b/README.md index 72bd1303c..9b8f96201 100644 --- a/README.md +++ b/README.md @@ -34,7 +34,7 @@ cp .env.example .env docker compose up -d --build ``` -Open `http://localhost:7000` when the containers are healthy. The first admin password is printed in `docker compose logs odysseus`. +Open `http://localhost:7011` when the containers are healthy. The first admin password is printed in `docker compose logs odysseus`. Native installs, GPU notes, Windows/macOS instructions, HTTPS, and configuration live in the [setup guide](website/setup.md). diff --git a/docker-compose.gpu-amd.yml b/docker-compose.gpu-amd.yml index ff6548fab..ce7e2d96d 100644 --- a/docker-compose.gpu-amd.yml +++ b/docker-compose.gpu-amd.yml @@ -59,7 +59,6 @@ services: - CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24} - ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1} - ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1} - - ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false} - ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1} - ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0} - ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0} diff --git a/docker-compose.gpu-nvidia.yml b/docker-compose.gpu-nvidia.yml index 53cd33699..08a9638b4 100644 --- a/docker-compose.gpu-nvidia.yml +++ b/docker-compose.gpu-nvidia.yml @@ -58,7 +58,6 @@ services: - CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24} - ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1} - ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1} - - ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false} - ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1} - ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0} - ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0} diff --git a/docker-compose.yml b/docker-compose.yml index 949167460..9e683482d 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -47,7 +47,6 @@ services: - CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24} - ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1} - ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1} - - ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false} - ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1} - ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0} - ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0} diff --git a/requirements.txt b/requirements.txt index 7b99707e5..a91cba857 100644 --- a/requirements.txt +++ b/requirements.txt @@ -11,7 +11,6 @@ pypdf pypdfium2 Pillow faster-whisper -PyPDF2 pdfplumber beautifulsoup4 charset-normalizer diff --git a/scripts/eval_followup_domain_continuity.py b/scripts/eval_followup_domain_continuity.py new file mode 100644 index 000000000..d35e7c270 --- /dev/null +++ b/scripts/eval_followup_domain_continuity.py @@ -0,0 +1,209 @@ +#!/usr/bin/env python3 +"""Evaluate follow-up domain continuity against an OpenAI-compatible API.""" +from __future__ import annotations + +import argparse +import asyncio +import json +import random +import sqlite3 +import sys +from collections import Counter +from pathlib import Path + +import httpx + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from src.tool_policy import ToolPolicy +from src.tool_routing_experiment import MODEL_CHOICE_MODE, select_experiment_inventory +from src.tool_schemas import FUNCTION_TOOL_SCHEMAS +from src.turn_contract import ( + _families_for_tool, + canonical_tool, + requested_capabilities, + resolve_full_inventory_contract, + resolve_turn_contract, +) + + +FAMILY_TOOL = { + "search_browser": "web_search", + "notes": "manage_notes", + "documents": "manage_documents", + "email": "search_emails", + "calendar": "manage_calendar", + "tasks": "manage_tasks", + "skills": "manage_skills", + "memory": "manage_memory", +} + +SUBJECTS = [ + "Rocket League", "Project Juniper", "Aurora Seven", "Kyoto Railway Museum", + "Framework Laptop", "Blue Harbor", "Atlas Report", "Orion Browser", + "Maple Invoice", "Cobalt Launch", "Sakura Booking", "Nimbus Checklist", +] + +FOLLOWUPS = [ + "I searched {subject} and cannot find it", + "{subject} is not showing for me", + "where is {subject} listed", + "I looked for {subject} but got nothing", + "why does {subject} not appear in the results", + "still no sign of {subject}", + "the result for {subject} seems to be missing", + "I tried again and {subject} is absent", +] + + +def history(subject: str, family: str) -> list[dict]: + tool = FAMILY_TOOL[family] + return [ + {"role": "user", "content": f"Find {subject}"}, + {"role": "assistant", "content": f"I found information about {subject}.", "metadata": { + "tool_events": [{"tool": tool, "exit_code": 0, "error": False}], + }}, + ] + + +def build_cases(count: int, seed: int) -> list[dict]: + rng = random.Random(seed) + families = tuple(FAMILY_TOOL) + cases = [] + for index in range(count): + source = families[index % len(families)] + subject = f"{rng.choice(SUBJECTS)} {index + 1}" + if index % 4: + prompt = rng.choice(FOLLOWUPS).format(subject=subject) + expected = source + kind = "continuation" + else: + target = families[(families.index(source) + 1 + index) % len(families)] + noun = { + "search_browser": "the web", "notes": "my notes", + "documents": "my documents", "email": "my email", + "calendar": "my calendar", "tasks": "my tasks", + "skills": "my skills", "memory": "my memories", + }[target] + prompt = f"Search {noun} for {subject} instead" + expected = target + kind = "explicit_switch" + cases.append({ + "id": index + 1, "kind": kind, "source": source, + "expected": expected, "subject": subject, "prompt": prompt, + }) + rng.shuffle(cases) + return cases + + +def contract_for(case: dict): + prior = history(case["subject"], case["source"]) + policy = ToolPolicy() + capabilities = requested_capabilities(case["prompt"], prior) + inventory = resolve_full_inventory_contract( + schemas=FUNCTION_TOOL_SCHEMAS, policy=policy, + ) + routed = resolve_turn_contract( + capabilities=capabilities, schemas=FUNCTION_TOOL_SCHEMAS, policy=policy, + ) + contract = select_experiment_inventory( + inventory, routed, prior, MODEL_CHOICE_MODE, user_text=case["prompt"], + ) + return prior, capabilities, contract + + +def endpoint_from_database(path: Path, session_id: str) -> tuple[str, str, dict]: + with sqlite3.connect(f"file:{path}?mode=ro", uri=True) as db: + row = db.execute( + "select endpoint_url, model, headers from sessions where id=?", (session_id,), + ).fetchone() + if not row: + raise SystemExit(f"Session {session_id} was not found in {path}") + return row[0], row[1], json.loads(row[2] or "{}") + + +async def run_case(client, semaphore, endpoint, model, headers, case): + prior, capabilities, contract = contract_for(case) + schemas = contract.schemas() + exposed = { + family for schema in schemas + for family in _families_for_tool(canonical_tool(schema["function"]["name"])) + } + allowed_families = {case["expected"]} + if case["expected"] == "email": + allowed_families.add("contacts") + result = {**case, "capabilities": sorted(capabilities), "exposed": sorted(exposed)} + result["contract_ok"] = bool(exposed) and exposed <= allowed_families + payload = { + "model": model, + "messages": [ + {"role": "system", "content": "Continue the conversation. Use an offered tool when evidence is needed."}, + *[{"role": row["role"], "content": row["content"]} for row in prior], + {"role": "user", "content": case["prompt"]}, + ], + "tools": schemas, + "tool_choice": "auto", + "temperature": 0.7, + "max_tokens": 96, + } + async with semaphore: + for attempt in range(4): + try: + response = await client.post(endpoint, headers=headers, json=payload) + if response.status_code == 429 and attempt < 3: + await asyncio.sleep(1.5 * (attempt + 1)) + continue + response.raise_for_status() + message = response.json()["choices"][0]["message"] + calls = message.get("tool_calls") or [] + tools = [canonical_tool(call["function"]["name"]) for call in calls] + called_families = sorted({ + family for tool in tools for family in _families_for_tool(tool) + }) + result.update({"tools": tools, "called_families": called_families}) + result["cross_domain"] = bool(set(called_families) - allowed_families) + result["expected_call"] = case["expected"] in called_families + return result + except Exception as exc: + if attempt == 3: + result["error"] = f"{type(exc).__name__}: {exc}" + return result + await asyncio.sleep(0.5 * (attempt + 1)) + + +async def main(args): + endpoint, model, headers = endpoint_from_database(args.database, args.session_id) + cases = build_cases(args.count, args.seed) + timeout = httpx.Timeout(45, connect=10) + semaphore = asyncio.Semaphore(args.concurrency) + async with httpx.AsyncClient(timeout=timeout) as client: + results = await asyncio.gather(*( + run_case(client, semaphore, endpoint, model, headers, case) + for case in cases + )) + summary = { + "count": len(results), + "contract_pass": sum(bool(row.get("contract_ok")) for row in results), + "cross_domain_calls": sum(bool(row.get("cross_domain")) for row in results), + "expected_tool_calls": sum(bool(row.get("expected_call")) for row in results), + "no_tool": sum(not row.get("tools") and not row.get("error") for row in results), + "errors": sum("error" in row for row in results), + "kinds": Counter(row["kind"] for row in results), + } + output = {"summary": summary, "results": results} + args.output.write_text(json.dumps(output, indent=2, default=dict) + "\n") + print(json.dumps(summary, indent=2, default=dict)) + raise SystemExit(1 if summary["contract_pass"] != len(results) or summary["cross_domain_calls"] else 0) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--count", type=int, default=500) + parser.add_argument("--concurrency", type=int, default=12) + parser.add_argument("--seed", type=int, default=20260921) + parser.add_argument("--database", type=Path, required=True) + parser.add_argument("--session-id", required=True) + parser.add_argument("--output", type=Path, required=True) + asyncio.run(main(parser.parse_args())) diff --git a/scripts/sft_email_overseer.py b/scripts/sft_email_overseer.py index 46621dffd..c38414271 100644 --- a/scripts/sft_email_overseer.py +++ b/scripts/sft_email_overseer.py @@ -205,7 +205,8 @@ def retarget_row(row: dict[str, Any], target_owner: str, uid_offset: int) -> dic try: new_uid = str(uid_offset + int(source_uid)) except ValueError: - new_uid = f"{uid_offset}{re.sub(r'\\W+', '', source_uid)[:8]}" + source_uid_suffix = re.sub(r"\W+", "", source_uid)[:8] + new_uid = f"{uid_offset}{source_uid_suffix}" cloned["owner"] = target_owner cloned["uid"] = new_uid diff --git a/services/memory/builtin_skills.py b/services/memory/builtin_skills.py index 46bce5c34..131673dbd 100644 --- a/services/memory/builtin_skills.py +++ b/services/memory/builtin_skills.py @@ -5,11 +5,13 @@ from __future__ import annotations from pathlib import Path from typing import Iterable +from src.constants import BUILTIN_SKILLS_DIR + from .skill_format import Skill from .skills import SkillsManager -_BUILTIN_ROOT = Path(__file__).resolve().parents[2] / "resources" / "skills" +_BUILTIN_ROOT = Path(BUILTIN_SKILLS_DIR) _SYNC_FIELDS = ( "name", "description", @@ -40,7 +42,7 @@ def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int Installation is safe before first-user setup because no owner identity is assigned and unauthenticated requests still cannot access skill routes. """ - existing = {row.get("name") for row in manager.load_all()} + existing = {row.get("name"): row for row in manager.load_all()} installed = 0 paths = sorted(_BUILTIN_ROOT.rglob("SKILL.md")) if _BUILTIN_ROOT.is_dir() else [] for path in paths: @@ -52,9 +54,8 @@ def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int # available immediately and never enter the user's audit queue. skill.status = "published" skill.confidence = 1.0 - existing_rows = [row for row in manager.load_all() if row.get("name") == skill.name] - if existing_rows: - row = existing_rows[0] + row = existing.get(skill.name) + if row: # Built-ins are immutable tracked assets. Synchronize updated # versions/procedures on startup while leaving usage counters in # their sidecar untouched. Older startup code could also stamp the @@ -64,11 +65,11 @@ def install_builtin_skills(manager: SkillsManager, owners: Iterable[str]) -> int skill.source = "builtin" desired = skill.to_dict() if any(row.get(field) != desired.get(field) for field in _SYNC_FIELDS): - manager._write_skill(skill) + manager.sync_builtin_skill(skill) continue skill.owner = "" skill.source = "builtin" - manager._write_skill(skill) - existing.add(skill.name) + manager.sync_builtin_skill(skill) + existing[skill.name] = skill.to_dict() installed += 1 return installed diff --git a/services/memory/skills.py b/services/memory/skills.py index 9d05f4798..a8882d5ad 100644 --- a/services/memory/skills.py +++ b/services/memory/skills.py @@ -226,6 +226,10 @@ class SkillsManager: sk.path = path return path + def sync_builtin_skill(self, skill: Skill) -> str: + """Persist a trusted built-in skill during startup synchronization.""" + return self._write_skill(skill) + def backfill_owner(self, primary_owner: str, valid_owners: Optional[set[str]] = None) -> int: """Assign legacy/unclaimed skill files to the primary owner. diff --git a/services/search/core.py b/services/search/core.py index fa0848712..dc97bfb61 100644 --- a/services/search/core.py +++ b/services/search/core.py @@ -226,6 +226,21 @@ def _meaningful_query_terms(query: str) -> list[str]: ] +# Leading function/auxiliary words carry no entity signal. They are kept out +# of _SEARCH_QUERY_FILLER (which gates overall query meaningfulness) and +# applied only to the document-cue entity test below, where taking the *first* +# surviving token as the entity otherwise picks "how"/"i"/"best" and rejects +# every genuinely relevant result. +_QUERY_FUNCTION_WORDS = frozenset({ + "how", "to", "i", "we", "you", "your", "my", "our", "me", "us", + "a", "an", "is", "are", "was", "were", "do", "does", "did", "can", + "could", "should", "would", "will", "get", "getting", "got", + "there", "here", "need", "needed", "want", "looking", "show", "give", + "help", "best", "good", "top", "recommended", "some", "it", "its", + "of", "in", "on", "at", "by", "or", "and", "be", "have", "has", +}) + + def _result_has_query_overlap(query: str, result: dict) -> bool: terms = _meaningful_query_terms(query) if not terms: @@ -247,7 +262,7 @@ def _result_has_query_overlap(query: str, result: dict) -> bool: "documentation", "docs", "pdf", "handbook", } if query_tokens & document_cues: - entity_fillers = _SEARCH_QUERY_FILLER | document_cues | { + entity_fillers = _SEARCH_QUERY_FILLER | document_cues | _QUERY_FUNCTION_WORDS | { "english", "operator", "owner", "owners", "user", "installation", } ordered_query_tokens = re.findall(r"[a-z0-9]+", str(query or "").lower()) @@ -258,7 +273,13 @@ def _result_has_query_overlap(query: str, result: dict) -> bool: # Product/manual lookups are especially vulnerable to homonyms. A # result matching only the generic product word and "manual" is not # evidence for the named brand/entity in the request. - if entity_terms and entity_terms[0] not in result_tokens: + # + # Test *any* entity term rather than specifically the first. Position + # does not identify the entity: "how to configure nginx docs" leads + # with a task verb, "best guide for sourdough" with a qualifier. A + # result naming none of the entity terms is still rejected, which is + # what keeps a Ford manual out of an IKEA BILLY lookup. + if entity_terms and not (set(entity_terms) & result_tokens): return False model_numbers = {token for token in ordered_query_tokens if token.isdigit()} # Temporal qualifiers are not product identifiers. In particular, @@ -296,12 +317,13 @@ def _result_has_query_overlap(query: str, result: dict) -> bool: def _filter_low_relevance_results(query: str, results: list[dict]) -> list[dict]: if not results: return [] - relevant = [result for result in results - if _result_matches_site_scope(query, result) - and _result_has_query_overlap(query, result)] - # Only reject a provider when it returned a fully off-topic page set. Mixed - # result pages are common; ranking can handle those. - return relevant if relevant else [] + scoped = [result for result in results if _result_matches_site_scope(query, result)] + relevant = [result for result in scoped if _result_has_query_overlap(query, result)] + # Relevance matching is intentionally conservative and cannot understand + # every inflection or language. Keep explicit site constraints strict, but + # let ranking handle a provider page when the heuristic rejects every + # otherwise in-scope result. + return relevant or scoped def _result_matches_site_scope(query: str, result: dict) -> bool: diff --git a/src/agent_loop.py b/src/agent_loop.py index e94facda4..0af073fbc 100644 --- a/src/agent_loop.py +++ b/src/agent_loop.py @@ -7672,7 +7672,7 @@ If a calendar create/update request lacks a required date, time, or target event "send_to_session": "- ```send_to_session``` — Send a message to another session. Line 1 = session_id, rest = message. Use for orchestrating work across sessions.", "search_chats": "- ```search_chats``` — Search past session transcripts for direct conversation evidence. Use when user asks 'did we discuss X?', 'find the conversation about Y', or when prior chat context is more appropriate than persistent memory.", "pipeline": "- ```pipeline``` — Run a multi-step AI pipeline. Args (JSON) with ordered steps, each specifying a model and prompt. Use for complex workflows.", - "ui_control": "- ```ui_control``` — Control the UI: toggle tools on/off, OPEN PANELS, open email reply drafts, switch models, change themes. Commands: `toggle on/off` (names: bash/shell, web/search, research, incognito, document_editor/documents), `open_panel ` (panels: documents, gallery, calendar/schedule, email, sessions, notes, memories/brain, skills, settings, theme, cookbook), `open_panel calendar month|week|year|agenda [YYYY-MM or YYYY-MM-DD]` (open calendar directly to a view/range), `open_email_reply ` (opens an email compose document pre-filled with body, DOES NOT send; use this for normal “write/draft a reply saying X” requests), `set_mode agent/chat`, `switch_model `, `set_theme `, `create_theme ` (optional key=val for advanced colors AND background effects: bgPattern=, bgEffectColor=#RRGGBB, bgEffectIntensity=, bgEffectSize=, frosted=true|false). \"open calendar\" / \"open schedule\" / \"open documents\" / \"open library\" / \"show gallery\" / \"open inbox\" / \"open notes\" / \"open theme\" / \"open cookbook\" all map to `open_panel `. Built-in theme presets: dark, light, midnight, paper, cyberpunk, retrowave, forest, ocean, ume, copper, terminal, organs, lavender, gpt, claude, cute. For any other vibe/name, use create_theme.", + "ui_control": "- ```ui_control``` — Control the UI: toggle tools on/off, OPEN PANELS, open email reply drafts, switch models, change themes. Commands: `toggle on/off` (names: bash/shell, web/search, research, incognito, document_editor/documents), `open_panel ` (panels: documents, gallery, calendar/schedule, email, sessions, notes, memories/brain, skills, settings, theme, cookbook), `open_panel calendar month|week|year|agenda [YYYY-MM or YYYY-MM-DD]` (open calendar directly to a view/range), `open_email_reply ` (opens an email compose document pre-filled with body, DOES NOT send; use this for normal “write/draft a reply saying X” requests), `set_mode agent/chat`, `switch_model `, `set_theme `, `create_theme ` (optional key=val for advanced colors AND background effects: bgPattern=, bgEffectColor=#RRGGBB, bgEffectIntensity=, bgEffectSize=, frosted=true|false). \"open calendar\" / \"open schedule\" / \"open documents\" / \"open library\" / \"show gallery\" / \"open inbox\" / \"open notes\" / \"open theme\" / \"open cookbook\" all map to `open_panel `. Built-in theme presets: dark, light, midnight, cyberpunk, retrowave, forest, ocean, ume, terminal, organs, gpt, claude, cute, eclipse, porcelain, arcade, blueprint, monolith, yoyo. For any other vibe/name, use create_theme.", "ask_user": "- ```ask_user``` — Ask the user a question when the task is genuinely ambiguous and the answer changes what you do next (pick an approach, confirm an assumption, choose a target). Args (JSON): {\"question\": \"...\", \"options\": [{\"label\": \"...\", \"description\": \"...\"?}, ...], \"multi\": false?}. 2-6 options. The user gets clickable buttons; calling this ENDS your turn and their choice comes back as your next message. For open-ended missing data such as an exact calendar date, include an \"Exact date\" option and ask the user to type the date; do not invent arbitrary choices. Prefer sensible defaults — only ask when you truly can't proceed well without their input.", "update_plan": "- ```update_plan``` — While executing an approved plan, write the plan back: tick steps done or revise them. Args (JSON): {\"plan\": \"- [x] done step\\n- [ ] next step\"}. Always pass the COMPLETE checklist, not a diff. Call it after finishing each step (mark it `- [x]`) and whenever the user asks to change the plan. The user's docked plan window updates live. Does nothing if there's no active plan.", "list_served_models": "- ```list_served_models``` — Show what the Cookbook (LLM-serving subsystem) is currently running. NO args. Use this for ANY 'what's running' / 'what's serving' / 'show my cookbook' / 'is anything up' query. DO NOT shell out (`ps aux`, `docker ps`, etc.) — this tool is the source of truth. Failed serve tasks include recent logs plus diagnosis/retry suggestions; use those suggestions to call `serve_model` again with an adjusted command when appropriate.", @@ -13695,6 +13695,76 @@ def _is_odysseus_qwen_native(model: str) -> bool: return bool(re.search(r"\bqwen3(?:\.?(?:6|8))-27b-(?:mlx|fp8)(?:\b|[-_/])", value)) +def _is_deepseek_flash_vision_model(model: str) -> bool: + """Recognize the provider's exact vision-capable Flash variant.""" + value = str(model or "").strip().lower().rstrip("/") + return value.rsplit("/", 1)[-1] == "deepseek-flash" + + +def _deepseek_flash_visual_continuation( + request_messages: Sequence[Mapping[str, Any]], + direct_user_text: str, +) -> Optional[list[dict]]: + """Flatten one post-tool visual turn for DeepSeek Flash. + + The hosted Flash vision path can reason over pixels and emit native tool + calls from a fresh multimodal request. It currently returns an empty, + length-terminated response when the image follows assistant/tool-call + history. Collapse only requests carrying explicit tool visual evidence; + ordinary text and later tool rounds retain their full history. + """ + newest_visual: Optional[Mapping[str, Any]] = None + for message in request_messages or (): + metadata = message.get("metadata") or {} + content = message.get("content") + if ( + message.get("role") == "user" + and isinstance(metadata, Mapping) + and metadata.get("source") == "tool visual evidence" + and isinstance(content, list) + and any( + isinstance(block, Mapping) and block.get("type") == "image_url" + for block in content + ) + ): + newest_visual = message + if newest_visual is None: + return None + + systems = [ + dict(message) + for message in request_messages or () + if message.get("role") == "system" + ] + visual_content = newest_visual.get("content") or [] + images = [ + dict(block) + for block in visual_content + if isinstance(block, Mapping) and block.get("type") == "image_url" + ] + if not images: + return None + task = str(direct_user_text or "").strip() + evidence_text = "\n".join( + str(block.get("text") or "").strip() + for block in visual_content + if isinstance(block, Mapping) + and block.get("type") == "text" + and str(block.get("text") or "").strip() + ) + instruction = ( + (f"{task}\n\n" if task else "") + + (f"{evidence_text}\n\n" if evidence_text else "") + + "The requested visual evidence is attached below. Analyze these pixels " + "directly and continue the task using downstream tools. Do not request " + "another inspection of this same view." + ) + return systems + [{ + "role": "user", + "content": [{"type": "text", "text": instruction}, *images], + }] + + def _ody_qwen_temperature_cap(temperature): """Force-cap odysseus-qwen3 sampling; the finetune destabilizes above 0.2. @@ -15663,9 +15733,30 @@ def _requested_post_edit_verification(text: str) -> bool: return False if _requested_verification_command(value): return True + if re.search( + r"\b(?:inspect|review|check|verify|read(?:\s+it)?\s+back)\b" + r".{0,100}\b(?:saved|written|created|output|file|artifact)\b", + value, + re.IGNORECASE | re.DOTALL, + ): + return True return bool(re.search( - r"\b(?:then|after(?:wards)?|and)\b.{0,100}\b(?:run|execute|test|verify|check|build|compile|lint)\b" - r"|\b(?:run|execute|test|verify|check|build|compile|lint)\b.{0,100}\b(?:after|once|when)\b", + r"\b(?:then|after(?:wards)?|and)\b.{0,100}\b(?:run|execute|test|verify|check|inspect|review|read(?:\s+it)?\s+back|build|compile|lint)\b" + r"|\b(?:run|execute|test|verify|check|inspect|review|read(?:\s+it)?\s+back|build|compile|lint)\b.{0,100}\b(?:after|once|when)\b", + value, + re.IGNORECASE | re.DOTALL, + )) + + +def _requested_artifact_readback(text: str) -> bool: + """Whether verification specifically asks to inspect the saved artifact.""" + + value = str(text or "") + return bool(re.search( + r"\b(?:inspect|review|check|verify|read(?:\s+it)?\s+back)\b" + r".{0,100}\b(?:saved|written|created|output|file|artifact)\b" + r"|\b(?:saved|written|created|output|file|artifact)\b" + r".{0,100}\b(?:inspect|review|check|verify|read(?:\s+it)?\s+back)\b", value, re.IGNORECASE | re.DOTALL, )) @@ -15703,6 +15794,32 @@ def _first_explicit_workspace_file(text: str) -> str: return _clean_file_edit_value(str(match.group("path") or "").strip().rstrip(".")) +def _read_file_block_path(content) -> str: + """Path argument of a read_file tool block, JSON args or bare text.""" + text = str(content or "").strip() + try: + args = json.loads(text) + if isinstance(args, dict): + return str(args.get("path") or "").strip() + except (TypeError, ValueError, json.JSONDecodeError): + pass + return text.splitlines()[0].strip() if text else "" + + +def _read_file_targets_artifact(content, target) -> bool: + """True when a read_file block reads the artifact awaiting verification. + + Reading the *input* named earlier in the same prompt must not satisfy a + request to verify the written output. + """ + if not target: + return False + path = _read_file_block_path(content) + if not path: + return False + return path == str(target) or Path(path).name == Path(str(target)).name + + def _explicit_workspace_files(text: str) -> list[str]: """Return concrete source/test paths named in a workspace request.""" paths: list[str] = [] @@ -20297,6 +20414,7 @@ async def stream_agent_loop( external_tool_schemas=external_tool_schemas, max_tokens=max_tokens, max_rounds=max_rounds, + max_tool_calls=max_tool_calls, temperature=temperature, ): yield chunk @@ -22847,6 +22965,16 @@ async def stream_agent_loop( ] + _declared_native_artifacts )) + # A forced read-back must target a declared *output*. + # _workspace_artifacts is in prompt order, so index 0 is the input for + # the ordinary "read /workspace/in/x, write /workspace/out/y, then + # check the saved file" shape. Declared required_artifacts are + # authoritative outputs; otherwise prefer the last named path, which is + # the deliverable in that phrasing, over the first. + _artifact_readback_target = ( + _declared_native_artifacts[0] if _declared_native_artifacts + else (_workspace_artifacts[-1] if _workspace_artifacts else None) + ) _artifact_creation_requested = bool( (workspace or _native_artifact_runtime) and _workspace_artifacts @@ -23156,6 +23284,16 @@ async def stream_agent_loop( "[agent-context] final trimmed request lost direct user turn; restoring it before provider call: %r", _last_user[:160], ) + _trimmed_visual_evidence = [ + message + for message in trimmed_messages + if ( + isinstance(message, dict) + and message.get("role") == "user" + and (message.get("metadata") or {}).get("source") + == "tool visual evidence" + ) + ] trimmed_messages = [ message for message in trimmed_messages if not ( @@ -23164,7 +23302,14 @@ async def stream_agent_loop( and (message.get("metadata") or {}).get("trusted") is False and (message.get("metadata") or {}).get("source") ) - ] + [{"role": "user", "content": _last_user}] + ] + [ + {"role": "user", "content": _last_user}, + # Keep the newest tool pixels after the restored task + # text. Provider sanitization merges these consecutive + # user turns into one final multimodal turn; hosted + # vision APIs may ignore images stranded in an older turn. + *_trimmed_visual_evidence, + ] after_trim_tokens = estimate_tokens(trimmed_messages) if after_trim_tokens < before_trim_tokens: logger.info( @@ -24019,6 +24164,7 @@ async def stream_agent_loop( _single_execution_bound = _request_forbids_execution_retry(_last_user) _execution_tool_attempts: dict[str, int] = {} _post_edit_verification_required = _requested_post_edit_verification(_last_user) + _artifact_readback_requested = _requested_artifact_readback(_last_user) _post_edit_verification_command = _requested_verification_command(_last_user) if _post_edit_verification_required and not _post_edit_verification_command and _tui_test_request: _post_edit_verification_command = _tui_local_fallback_shell_command( @@ -25265,7 +25411,19 @@ async def stream_agent_loop( candidate_model, state["messages"], ) - state["request_messages"] = request_messages + deepseek_visual_messages = None + if _is_deepseek_flash_vision_model(candidate_model): + deepseek_visual_messages = _deepseek_flash_visual_continuation( + request_messages, + _last_user, + ) + if deepseek_visual_messages is not None: + request_messages = deepseek_visual_messages + logger.info( + "[agent] flattened DeepSeek Flash post-tool visual " + "continuation and suppressed redundant inspect_media" + ) + state["request_messages"] = request_messages _last_route_request_messages = request_messages state["context_length"] = _route_context_lengths.get( (candidate_url, candidate_model), @@ -25274,6 +25432,12 @@ async def stream_agent_loop( _last_route_context_length = state["context_length"] run_security.observe_messages(request_messages) candidate_tools = _tool_schemas_for_route(state) + if deepseek_visual_messages is not None: + candidate_tools = [ + schema + for schema in candidate_tools or () + if schema.get("function", {}).get("name") != "inspect_media" + ] state["tools"] = candidate_tools from src.generation_budget import fit_output_token_budget @@ -27170,7 +27334,31 @@ async def stream_agent_loop( native_tool_calls = [] used_native = False logger.info("[agent] normalized inspection follow-up to one edit_file call") - if ( + elif ( + _artifact_readback_requested + and _post_effectful_mutation_done + and not _post_edit_verification_completed + and not _post_edit_verification_force_attempted + and _artifact_readback_target + ): + # The user explicitly asked to inspect the saved artifact. Once a + # write succeeds, normalize one bounded read-back rather than + # letting a weak router reopen source-media inspection forever. + # Chained onto the preceding branches: an already-normalized + # authorized edit must not be overwritten by this read. + tool_blocks = [ToolBlock( + "read_file", + json.dumps({"path": _artifact_readback_target}), + )] + converted_calls = [] + native_tool_calls = [] + used_native = False + _post_edit_verification_force_attempted = True + logger.info( + "[agent] normalized post-edit artifact verification to read_file: %s", + _artifact_readback_target, + ) + elif ( _post_edit_verification_nudge_sent and (_post_effectful_mutation_done or _inspection_edit_completed or _file_creation_completed) and not _post_edit_verification_completed @@ -29453,9 +29641,13 @@ async def stream_agent_loop( "role": "system", "content": ( "The requested file edit succeeded, but the user also asked " - "for verification. Do that now with one concrete tool call " - "using the requested command (host_shell), then summarize. " - "Do not stop after the edit." + "for verification. " + + ( + "Read the saved output artifact now with read_file, then summarize. " + if _artifact_readback_requested + else "Do that now with one concrete tool call using the requested command (host_shell), then summarize. " + ) + + "Do not stop after the edit." ), }) yield f'data: {json.dumps({"type": "agent_step", "round": round_num + 1})}\n\n' @@ -34531,7 +34723,7 @@ async def stream_agent_loop( ] _workspace_read_requires_mutation = True if ( - block.tool_type in {"host_shell", "bash", "python"} + block.tool_type in {"host_shell", "bash", "python", "read_file"} and tool_result_is_successful(result) and ( _post_edit_verification_nudge_sent @@ -34539,6 +34731,20 @@ async def stream_agent_loop( ) and ( not _post_edit_verification_required + or ( + block.tool_type == "read_file" + and _artifact_readback_requested + and _post_effectful_mutation_done + # Only the artifact under verification counts. When no + # target could be resolved, fall back to the previous + # any-read behaviour so the turn cannot deadlock. + and ( + not _artifact_readback_target + or _read_file_targets_artifact( + block.content, _artifact_readback_target + ) + ) + ) or ( _tui_test_request and command_is_test(block.content) @@ -35033,9 +35239,13 @@ async def stream_agent_loop( "role": "system", "content": ( "The requested file edit succeeded, but the user also asked " - "for verification. Do that now with one concrete tool call " - "using the requested command (host_shell), then summarize. " - "Do not stop after the edit." + "for verification. " + + ( + "Read the saved output artifact now with read_file, then summarize. " + if _artifact_readback_requested + else "Do that now with one concrete tool call using the requested command (host_shell), then summarize. " + ) + + "Do not stop after the edit." ), }) yield f'data: {json.dumps({"type": "agent_step", "round": round_num + 1})}\n\n' diff --git a/src/ai_interaction.py b/src/ai_interaction.py index ed2496da8..156612011 100644 --- a/src/ai_interaction.py +++ b/src/ai_interaction.py @@ -695,7 +695,7 @@ async def do_ui_control(content: str, session_id: Optional[str] = None, owner: O toggle — Toggle a setting (web, bash, rag, research, incognito, document_editor) set_mode — Switch between agent and chat mode switch_model — Change the model for the current session - set_theme — Apply a built-in theme preset (dark, light, midnight, paper, cyberpunk, retrowave, forest, ocean, ume, copper, terminal, organs, lavender, gpt, claude, cute) + set_theme — Apply a built-in theme preset (dark, light, midnight, cyberpunk, retrowave, forest, ocean, ume, terminal, organs, gpt, claude, cute, eclipse, porcelain, arcade, blueprint, monolith, yoyo) create_theme [key=val ...] — Create custom theme. Optional key=val: advanced color overrides AND background effects: bgPattern=, bgEffectColor=#RRGGBB, bgEffectIntensity=, bgEffectSize=, frosted=true|false get_theme — Return the last server-synchronized theme for this user open_panel [view] — Open a panel; Cookbook views are download/models, launch/serve, active/running, dependencies, settings @@ -798,9 +798,9 @@ async def do_ui_control(content: str, session_id: Optional[str] = None, owner: O # Also check user's custom themes stored in prefs. # Must match the THEMES keys in static/js/theme.js. known_presets = [ - "dark", "light", "midnight", "paper", "cyberpunk", "retrowave", - "forest", "ocean", "ume", "copper", "terminal", "organs", - "lavender", "gpt", "claude", "cute", + "dark", "light", "midnight", "cyberpunk", "retrowave", "forest", + "ocean", "ume", "terminal", "organs", "gpt", "claude", "cute", + "eclipse", "porcelain", "arcade", "blueprint", "monolith", "yoyo", ] custom_themes = {} try: diff --git a/src/chat_helpers.py b/src/chat_helpers.py index 4fc675716..61d5db20b 100644 --- a/src/chat_helpers.py +++ b/src/chat_helpers.py @@ -50,6 +50,9 @@ _VISION_MODEL_KEYWORDS = ( # Qwen3.5 is a natively multimodal family even when a served-model alias # omits the traditional "VL" suffix (for example qwen35-9b-base-native). "qwen3.5", "qwen3_5", "qwen35", + # The hosted Flash alias accepts images despite lacking a vision/VL suffix. + # Keep this exact: deepseek-v4-pro on the same provider is text-only. + "deepseek-flash", # multimodal families whose names don't contain "vision"/"vl" but DO accept # images — without these the image is silently dropped for common Ollama tags # like gemma3:4b or gemma4:12b (issue #1274). Gemma 3/4 (4b+), Llama 4 (all), diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 4dcd18a0b..731444f2f 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -55,6 +55,9 @@ NATIVE_ROUND_LIMIT = 64 # and interaction are separate observable actions. INTERACTIVE_TOOL_CALL_LIMIT = 18 INTERACTIVE_BROWSER_TOOL_CALL_LIMIT = 30 +# "Unlimited" as a comparable int: every call site tests `calls < limit`, so a +# sentinel avoids threading an Optional through the whole preview loop. +UNLIMITED_TOOL_CALL_LIMIT = 1_000_000 INTERACTIVE_ROUND_LIMIT = 8 # Multi-record research tasks routinely need several search/fetch/inspection # pairs before an artifact can be grounded. Preserve twelve calls for writing, @@ -1468,15 +1471,41 @@ def standalone_social_turn(text): def interactive_execution_limit(max_rounds): - """Bound interactive turns independently of long-running native jobs.""" + """Honor the WebUI agent-step setting for the compact preview loop. + + A configured finite budget is the user's explicit instruction and is + honored up to the same 200 ceiling the settings endpoint enforces. + INTERACTIVE_ROUND_LIMIT remains the fallback when no budget is resolvable + (adaptive ``None`` mode or a malformed value), so a turn still terminates. + """ if max_rounds is None: return INTERACTIVE_ROUND_LIMIT try: - return max(1, min(int(max_rounds), INTERACTIVE_ROUND_LIMIT)) + return max(1, min(int(max_rounds), 200)) except (TypeError, ValueError): return INTERACTIVE_ROUND_LIMIT +def interactive_tool_call_limit(max_tool_calls, *, browser_offered=False): + """Honor the configured agent tool-call budget; 0 means unlimited. + + Matches the main agent loop, which treats ``max_tool_calls <= 0`` as + unbounded. The INTERACTIVE_* constants remain the fallback for a + malformed value. + """ + default = ( + INTERACTIVE_BROWSER_TOOL_CALL_LIMIT if browser_offered + else INTERACTIVE_TOOL_CALL_LIMIT + ) + try: + budget = int(max_tool_calls) + except (TypeError, ValueError): + return default + if budget <= 0: + return UNLIMITED_TOOL_CALL_LIMIT + return budget + + def runtime_required_artifacts(user_text, client_runtime_context): """Use runner-declared outputs, falling back to prompt inference.""" context = client_runtime_context if isinstance(client_runtime_context, dict) else {} @@ -4314,6 +4343,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac history_session=None, external_untrusted_context_seen=False, active_document=None, active_email=None, workspace=None, client_runtime_context=None, max_tokens=768, max_rounds=8, + max_tool_calls=0, external_tool_schemas=None, temperature=0.0, **ignored): from src.generation_sampling import validate_temperature @@ -4577,10 +4607,12 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac 'Open the document you want reviewed, then ask for inline suggestions again.' ) round_limit = interactive_execution_limit(max_rounds) - tool_call_limit = ( - INTERACTIVE_BROWSER_TOOL_CALL_LIMIT - if any(canonical(schema['function']['name']) == 'private_browser' for schema in offered) - else INTERACTIVE_TOOL_CALL_LIMIT + tool_call_limit = interactive_tool_call_limit( + max_tool_calls, + browser_offered=any( + canonical(schema['function']['name']) == 'private_browser' + for schema in offered + ), ) if native_workspace_enabled: try: @@ -4929,6 +4961,11 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if content and not prior_summary_answer: yield event({'delta': content}) proposed = [pending[i] for i in sorted(pending)] + # A lead-in emitted before a tool call is live progress, not + # part of the terminal answer. Replace that draft when the + # eventual synthesis begins instead of concatenating both. + if proposed and streamed_round_text: + replace_streamed_draft_on_finish = True unexecutable_dsml_completion = False if not proposed and 'DSML' in content: offered_by_canonical = { diff --git a/src/constants.py b/src/constants.py index d5573fa5e..2b4919d3d 100644 --- a/src/constants.py +++ b/src/constants.py @@ -6,6 +6,7 @@ import subprocess from src.runtime_paths import get_app_root, get_default_data_dir APP_VERSION = "1.0.3" +BUILTIN_SKILLS_DIR = os.path.join(get_app_root(), "resources", "skills") # Identifies the private maintainer-preview build without changing the public # application semver used by release and readiness checks. Keep the API/UI # value tied to HARNESS_VERSION so a version bump cannot leave the running diff --git a/src/embeddings.py b/src/embeddings.py index 19cd09e43..a56e8666e 100644 --- a/src/embeddings.py +++ b/src/embeddings.py @@ -178,6 +178,21 @@ class FastEmbedClient: except Exception as _e: logger.debug("embedding cache symlink-heal skipped: %s", _e) kwargs = {"model_name": self.model, "cache_dir": cache_dir} + # Isolated evaluation and worker fleets can run many Odysseus + # processes on one host. FastEmbed otherwise lets ONNX Runtime size + # a thread pool from the whole machine for every process, which can + # create hundreds of threads per worker and starve inference. Keep + # the existing default for normal installs, but allow operators to + # bound that pool explicitly. + raw_threads = os.getenv("FASTEMBED_THREADS", "").strip() + if raw_threads: + try: + threads = int(raw_threads) + except ValueError as exc: + raise ValueError("FASTEMBED_THREADS must be an integer") from exc + if not 1 <= threads <= 256: + raise ValueError("FASTEMBED_THREADS must be between 1 and 256") + kwargs["threads"] = threads self._embedding = TextEmbedding(**kwargs) self._dim: Optional[int] = None self.url = "local://fastembed" diff --git a/src/tool_execution.py b/src/tool_execution.py index 2cdc8fd80..57707a693 100644 --- a/src/tool_execution.py +++ b/src/tool_execution.py @@ -737,7 +737,7 @@ _SENSITIVE_BASENAMES: set[str] = { _SENSITIVE_FILE_PATTERNS: tuple[str, ...] = ( "authorized_keys", "id_rsa", "id_ed25519", "id_ecdsa", - "known_hosts", + "known_hosts", "auth.json", "app.db", "settings.json", ) # Case-folded views used for matching. On a case-insensitive filesystem diff --git a/src/tool_routing_experiment.py b/src/tool_routing_experiment.py index a90249dcc..6dd21aa00 100644 --- a/src/tool_routing_experiment.py +++ b/src/tool_routing_experiment.py @@ -76,7 +76,13 @@ def select_experiment_inventory(inventory, routed, history, mode, *, user_text=' and routed.required_read_operation is None and any(canonical_tool(name) == 'edit_image' for name in routed.required) ) - if not explicit_image_edit: + if not explicit_image_edit and not ( + mode == MODEL_CHOICE_MODE and families + and routed.required_read_operation is None + ): + # A concrete current-turn route owns the inventory. Reintroducing + # previously used families here lets an explicit domain switch retain + # stale authority and lets a resolved follow-up drift into shell. families.update(recently_executed_families( history, user_turns=6, maximum=3, include_failed_attempts=mode == MODEL_CHOICE_MODE, diff --git a/src/tool_schemas.py b/src/tool_schemas.py index 88278d4d6..beb3c5ccf 100644 --- a/src/tool_schemas.py +++ b/src/tool_schemas.py @@ -895,7 +895,7 @@ FUNCTION_TOOL_SCHEMAS = [ "type": "function", "function": { "name": "ui_control", - "description": "Control the user interface. Actions: toggle (turn tools on/off), open_panel (open a modal: documents/library, gallery, calendar/schedule, email, sessions, notes, memories/brain, skills, settings, theme, cookbook; calendar supports month/week/year/agenda plus a date; Cookbook supports models/download, launch/serve, active/running, dependencies, and settings views), open_email_reply (legacy UI-only reply opener; prefer email MCP draft_email_reply for assistant-written reply drafts so a normal document-backed email draft is created), set_mode, switch_model, set_theme (built-in presets: dark, light, midnight, paper, cyberpunk, retrowave, forest, ocean, ume, copper, terminal, organs, lavender, gpt, claude, cute), create_theme (CREATE any custom theme with a name + colors object — pick distinctive, evocative hex colors that match the requested aesthetic, NOT generic defaults. The theme auto-applies after creation), get_theme, and get_toggles. When a user asks for ANY theme not in the built-in preset list, ALWAYS use create_theme.", + "description": "Control the user interface. Actions: toggle (turn tools on/off), open_panel (open a modal: documents/library, gallery, calendar/schedule, email, sessions, notes, memories/brain, skills, settings, theme, cookbook; calendar supports month/week/year/agenda plus a date; Cookbook supports models/download, launch/serve, active/running, dependencies, and settings views), open_email_reply (legacy UI-only reply opener; prefer email MCP draft_email_reply for assistant-written reply drafts so a normal document-backed email draft is created), set_mode, switch_model, set_theme (built-in presets: dark, light, midnight, cyberpunk, retrowave, forest, ocean, ume, terminal, organs, gpt, claude, cute, eclipse, porcelain, arcade, blueprint, monolith, yoyo), create_theme (CREATE any custom theme with a name + colors object — pick distinctive, evocative hex colors that match the requested aesthetic, NOT generic defaults. The theme auto-applies after creation), get_theme, and get_toggles. When a user asks for ANY theme not in the built-in preset list, ALWAYS use create_theme.", "parameters": { "type": "object", "properties": { diff --git a/src/turn_contract.py b/src/turn_contract.py index aeec06bd8..89467370b 100644 --- a/src/turn_contract.py +++ b/src/turn_contract.py @@ -4078,6 +4078,82 @@ def recently_executed_families(history: Iterable, *, user_turns: int = 6, return tuple(found) +_CONTINUITY_STOP_WORDS = frozenset({ + "a", "about", "an", "and", "are", "at", "be", "but", "can", "could", + "did", "do", "does", "for", "from", "get", "have", "how", "i", "in", + "is", "it", "look", "me", "my", "not", "of", "on", "or", "please", + "search", "searched", "searching", "see", "show", "that", "the", "them", + "there", "these", "this", "those", "to", "u", "was", "what", "when", + "where", "which", "why", "with", "you", "your", "whats", "what's", + "cant", "can't", "cannot", "dont", "don't", "doesnt", "doesn't", +}) + + +def _subject_tokens(value: object) -> frozenset[str]: + """Return content-bearing tokens for conversation-subject continuity.""" + return frozenset( + token for token in re.findall(r"[\w'-]+", str(value or "").casefold()) + if len(token) > 2 and token not in _CONTINUITY_STOP_WORDS + ) + + +def _immediate_prior_user_subject_tokens(history: Iterable) -> frozenset[str]: + rows = tuple(history or ()) + seen_assistant = False + for row in reversed(rows): + role = row.get("role") if isinstance(row, dict) else getattr(row, "role", "") + if role == "assistant" and not seen_assistant: + seen_assistant = True + continue + if seen_assistant and role == "user": + content = row.get("content", "") if isinstance(row, dict) else getattr(row, "content", "") + return _subject_tokens(content) + return frozenset() + + +def immediately_established_family(message: str, history: Iterable) -> str | None: + """Resolve an elliptical follow-up against the immediately proven domain. + + Tool events provide the typed domain; subject-token overlap only determines + whether the new sentence continues that turn. This deliberately does not + infer authority from older turns or from model prose. + """ + rows = tuple(history or ()) + assistant_index = None + families: set[str] = set() + for index in range(len(rows) - 1, -1, -1): + row = rows[index] + role = row.get("role") if isinstance(row, dict) else getattr(row, "role", "") + if role != "assistant": + continue + assistant_index = index + metadata = row.get("metadata") if isinstance(row, dict) else getattr(row, "metadata", None) + if isinstance(metadata, str): + try: + metadata = json.loads(metadata) + except (TypeError, json.JSONDecodeError): + metadata = {} + for event in (metadata or {}).get("tool_events") or (): + if event.get("error") is True or event.get("exit_code") not in (None, 0): + continue + families.update(_families_for_tool(canonical_tool(event.get("tool", "")))) + break + if assistant_index is None or len(families) != 1: + return None + + prior_user_text = "" + for row in reversed(rows[:assistant_index]): + role = row.get("role") if isinstance(row, dict) else getattr(row, "role", "") + if role == "user": + prior_user_text = row.get("content", "") if isinstance(row, dict) else getattr(row, "content", "") + break + if not prior_user_text: + return None + if _subject_tokens(message) & _subject_tokens(prior_user_text): + return next(iter(families)) + return None + + def recently_read_gallery(history: Iterable, *, user_turns: int = 6) -> bool: """Whether a recent successful app_api call established gallery context.""" turns = 0 @@ -4140,6 +4216,58 @@ def recently_read_gallery(history: Iterable, *, user_turns: int = 4) -> bool: return False +# Personal-data product nouns. A broad-briefing phrase ("what's new", +# "give me an update", "news") must not out-rank these: the user is asking +# about their own store, not the open Web. Scoped to a first-person +# possessive so open-web subjects that merely borrow a product noun +# ("the latest events in Kyiv") keep their Web route. +_PERSONAL_STORE_NOUNS = ( + r"(?:e?mails?|inbox|mailbox|calendar|calender|events?|appointments?|" + r"meetings?|agenda|notes?|checklists?|tasks?|todos?|documents?|docs?|" + r"memor(?:y|ies)|contacts?|skills?|sessions?|chats?|conversations?)" +) +_PERSONAL_STORE_SUBJECT = re.compile( + rf"\b(?:my|our)\b(?:\s+\w+){{0,2}}\s+{_PERSONAL_STORE_NOUNS}\b|" + rf"\b(?:inbox|mailbox)\b", + re.I, +) + + +_PERSONAL_STORE_FAMILY = ( + (("email", "emails", "mail", "mails", "inbox", "mailbox"), "email"), + (("calendar", "calender", "event", "events", "appointment", "appointments", + "meeting", "meetings", "agenda"), "calendar"), + (("note", "notes", "checklist", "checklists"), "notes"), + (("task", "tasks", "todo", "todos"), "tasks"), + (("document", "documents", "doc", "docs"), "documents"), + (("memory", "memories"), "memory"), + (("contact", "contacts"), "contacts"), + (("skill", "skills"), "skills"), + (("session", "sessions", "chat", "chats", "conversation", "conversations"), + "sessions"), +) + + +def names_personal_store(message: str) -> bool: + """True when the request names the user's own data store.""" + return bool(_PERSONAL_STORE_SUBJECT.search(str(message or ""))) + + +def personal_store_families(message: str) -> frozenset[str]: + """Families for the user's own stores named in a broad-briefing request. + + A briefing phrase must resolve to the named store rather than falling + through to an empty inventory, which would offer no tools at all. + """ + families: set[str] = set() + for match in _PERSONAL_STORE_SUBJECT.finditer(str(message or "")): + matched = match.group(0).lower() + for nouns, family in _PERSONAL_STORE_FAMILY: + if any(re.search(rf"\b{noun}\b", matched) for noun in nouns): + families.add(family) + return frozenset(families) + + def broad_web_briefing_request(message: str) -> bool: """Recognize requests that need broad, current, multi-source Web evidence.""" text = _normalize_request_lead(message) @@ -4180,6 +4308,18 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum if lead := _CONVERSATIONAL_ACTION_LEAD.fullmatch(text): text = lead["request"].strip() history = tuple(history) + repeated_subject = _subject_tokens(text) & _immediate_prior_user_subject_tokens(history) + scope_text = " ".join( + token for token in re.findall(r"[\w'-]+", text) + if token.casefold() not in repeated_subject + ) + newly_named_families = { + family for family, pattern in _FAMILY_WORDS.items() + if re.search(pattern, scope_text, re.I) + } + established_family = immediately_established_family(text, history) + if established_family and not newly_named_families: + return frozenset({established_family}) concrete_urls = re.findall(r"\bhttps?://[^\s<>\"']+", raw_text, re.I) workspace_media = re.search( r"(?:file://)?/workspace/[^\s`\"']+\." @@ -4250,6 +4390,9 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum broad_web_briefing_request(text) and not re.search(r"\b(?:research|investigate|deep[ -]?dive)\b", text, re.I) ): + _personal = personal_store_families(text) + if _personal: + return _personal return frozenset({"search_browser"}) if re.search(r"\b(?:web_search|web_fetch)\b", raw_text, re.I): # Explicit native-tool requests are stronger than incidental domain @@ -4267,7 +4410,10 @@ def requested_capabilities(message: str, history: Iterable = (), *, active_docum if ( re.search(r"\b(?:latest|recent|current|today(?:'s)?)\b", text, re.I) and re.search(r"\b(?:info(?:rmation)?|news|nees|updates?)\b", text, re.I) + and not names_personal_store(text) ): + # A named personal store out-ranks the broad-briefing route; the + # guard above lets those fall through to the family grammar. # Broad current-information requests still require live Web evidence. # Keep the common ``nees`` typo because a missed route leaves the model # with no way to answer and encourages it to ask unnecessary questions. @@ -5792,7 +5938,7 @@ def resolve_turn_contract(*, capabilities: Iterable[str], schemas: Iterable[dict if ( message is not None and selected_tools is not None - and set(selected_tools) & {"web_search", "web_fetch"} + and selected & {"web_search", "web_fetch"} ): # Browser is not core. It is a bounded recovery capability for a web # turn when static search/fetch cannot read the named site. diff --git a/static/app.js b/static/app.js index 639b4f468..0a562ac73 100644 --- a/static/app.js +++ b/static/app.js @@ -10,7 +10,7 @@ import modelsModule from './js/models.js'; import ragModule from './js/rag.js'; import presetsModule from './js/presets.js?v=20260908personaname1'; import searchModule from './js/search.js'; -import chatModule from './js/chat.js?v=20260916largetoolscroll2'; +import chatModule from './js/chat.js?v=20260917toolttft1'; import compareModule from './js/compare/index.js?v=20260909mobilepaneaddscroll1'; import documentModule from './js/document.js?v=20260916docctx2'; import searchChatModule from './js/search-chat.js'; @@ -22,7 +22,7 @@ import { settleSessionHydration } from './js/startupShell.js'; import markdownModule from './js/markdown.js'; -import chatRenderer from './js/chatRenderer.js?v=20260914pdfstrip1'; +import chatRenderer from './js/chatRenderer.js?v=20260914metricssummary1'; // Keep this specifier identical to every consumer (especially chat.js). // Different query strings create separate ES-module instances with separate // current-session state, so the picker can display one model while chat sends diff --git a/static/index.html b/static/index.html index 9aad22f7b..808f78340 100644 --- a/static/index.html +++ b/static/index.html @@ -2614,7 +2614,7 @@ - + diff --git a/static/js/admin.js b/static/js/admin.js index 9c144d481..0dccd8e68 100644 --- a/static/js/admin.js +++ b/static/js/admin.js @@ -2,7 +2,7 @@ // Admin-only: users, endpoints, MCP, RAG, embeddings, tokens, webhooks, features import uiModule from './ui.js?v=20260916largetoolscroll1'; -import settingsModule from './settings.js?v=20260909defaultmodelfix1'; +import settingsModule from './settings.js?v=20260912writingstyle3'; import { providerLogo, providerLogoFromUrl } from './providers.js'; import { sortModelObjects } from './modelSort.js'; import { PROVIDER_DEVICE_FLOWS, formatDeviceFlowError, runProviderDeviceFlow } from './providerDeviceFlow.js'; diff --git a/static/js/chat.js b/static/js/chat.js index 61b49fd9d..e8481d59e 100644 --- a/static/js/chat.js +++ b/static/js/chat.js @@ -9,7 +9,7 @@ import Storage from './storage.js'; import uiModule from './ui.js?v=20260916largetoolscroll1'; import sessionModule from './sessions.js'; import chatRenderer, { renderToolIcon } from './chatRenderer.js?v=20260914metricssummary1'; -import chatStream from './chatStream.js?v=20260913richdiff1'; +import chatStream from './chatStream.js?v=20260914pdfstrip1'; import { addAITTSButton } from './tts-ai.js'; import markdownModule from './markdown.js'; import spinnerModule from './spinner.js'; diff --git a/static/js/chatRenderer.js b/static/js/chatRenderer.js index b877132bf..114cc0365 100644 --- a/static/js/chatRenderer.js +++ b/static/js/chatRenderer.js @@ -6,7 +6,7 @@ import markdownModule from './markdown.js'; import { svgifyEmoji } from './markdown.js'; import { addAITTSButton } from './tts-ai.js'; import { providerLogo, providerLabel } from './providers.js'; -import settingsModule from './settings.js?v=20260909defaultmodelfix1'; +import settingsModule from './settings.js?v=20260912writingstyle3'; import spinnerModule from './spinner.js'; import { bindMenuDismiss } from './escMenuStack.js'; import { loadPanel } from './panels.js?v=20260909movepicklayer1'; @@ -1800,7 +1800,7 @@ function _activateEntityAnchor(e, forcedAnchor = null) { } else if (panel === 'skills') { document.getElementById('tool-skills-btn')?.click(); } else if (panel === 'research') { - import('./research/panel.js?v=20260911researchmenu1').then(mod => { + import('./research/panel.js?v=20260913researchrailerrors1').then(mod => { const open = mod.openPanel || (mod.default && mod.default.openPanel); if (open) open(); }).catch(() => {}); @@ -1855,7 +1855,7 @@ function _activateEntityAnchor(e, forcedAnchor = null) { if (open) open(id); }).catch(() => {}); } else if (kind === 'note') { - import('./notes.js?v=20260910drawmerge1').then(mod => { + import('./notes.js?v=20260911notesselectioncancel1').then(mod => { const open = mod.openNote || (mod.default && mod.default.openNote); if (open) open(id); try { @@ -1905,7 +1905,7 @@ function _activateEntityAnchor(e, forcedAnchor = null) { if (open) open(id); }).catch(() => {}); } else if (kind === 'research') { - import('./research/panel.js?v=20260911researchmenu1').then(mod => { + import('./research/panel.js?v=20260913researchrailerrors1').then(mod => { const open = mod.openPanel || (mod.default && mod.default.openPanel); if (open) open(id); }).catch(() => {}); @@ -2659,8 +2659,8 @@ export function displayMetrics(messageElement, metrics) {
Message stats
Model${model.split('/').pop()}
-
Input · all rounds${inputTokens.toLocaleString()} tokens${isReal ? '' : '~'}
- ${injectedTokens != null ? `
Injected · first request${Number(injectedTokens).toLocaleString()} tokens
` : ''} +
Input${inputTokens.toLocaleString()} tokens${isReal ? '' : '~'}
+ ${injectedTokens != null ? `
Injected${Number(injectedTokens).toLocaleString()} tokens
` : ''}
Output${outputTokens.toLocaleString()} tokens${isReal ? '' : '~'}
Total${totalTok.toLocaleString()} tokens
diff --git a/static/js/chatStream.js b/static/js/chatStream.js index 72e97973c..4981ac34d 100644 --- a/static/js/chatStream.js +++ b/static/js/chatStream.js @@ -186,7 +186,7 @@ export function handleUIControl(uiData) { if (fn) fn(); }).catch(function(){}); } else if (panel === 'calendar') { - import('./calendar.js?v=20260914emailsource9').then(function(mod) { + import('./calendar.js?v=20260914emailsource11').then(function(mod) { var viewFn = mod.openCalendarView || (mod.default && mod.default.openCalendarView); var fn = mod.openCalendar || (mod.default && mod.default.openCalendar); if (viewFn && (uiData.view || uiData.target_date)) viewFn(uiData.view || 'month', uiData.target_date || ''); @@ -208,7 +208,7 @@ export function handleUIControl(uiData) { if (fn) fn(uiData.view ? { tab: uiData.view } : undefined); }).catch(function(){}); } else if (panel === 'notes') { - import('./notes.js?v=20260910drawmerge1').then(function(mod) { + import('./notes.js?v=20260911notesselectioncancel1').then(function(mod) { var fn = mod.openPanel || mod.openNotes || (mod.default && (mod.default.openPanel || mod.default.openNotes)); if (fn) fn(); }).catch(function(){}); @@ -229,7 +229,7 @@ export function handleUIControl(uiData) { var ids = { memories: 'tool-memory-btn', skills: 'tool-skills-btn', settings: 'open-settings-btn' }; var btn = document.getElementById(ids[panel]); if (panel === 'settings') { - import('./settings.js?v=20260909defaultmodelfix1').then(function(mod) { + import('./settings.js?v=20260912writingstyle3').then(function(mod) { var fn = mod.open || (mod.default && mod.default.open); if (fn) fn(); else if (btn) btn.click(); diff --git a/static/js/compare/stream.js b/static/js/compare/stream.js index 10ebc2967..4ea5c4e74 100644 --- a/static/js/compare/stream.js +++ b/static/js/compare/stream.js @@ -1,7 +1,7 @@ // compare/stream.js — SSE streaming to panes import state from './state.js'; import { addFinishBadge } from './vote.js?v=20260828resendcaldrag1'; -import { getModelCost, renderAskUserCard, safeDisplayImageSrc } from '../chatRenderer.js?v=20260910streamlinks2'; +import { getModelCost, renderAskUserCard, safeDisplayImageSrc } from '../chatRenderer.js?v=20260914metricssummary1'; import markdownModule from '../markdown.js'; import spinnerModule from '../spinner.js'; import uiModule from '../ui.js?v=20260916largetoolscroll1'; diff --git a/static/js/compare/vote.js b/static/js/compare/vote.js index 1dd53b305..cece556e8 100644 --- a/static/js/compare/vote.js +++ b/static/js/compare/vote.js @@ -2,7 +2,7 @@ import Storage from '../storage.js'; import state from './state.js'; import { _modelDisplayNames } from './models.js'; -import { getModelCost } from '../chatRenderer.js?v=20260910streamlinks2'; +import { getModelCost } from '../chatRenderer.js?v=20260914metricssummary1'; import uiModule from '../ui.js?v=20260916largetoolscroll1'; import { VOTES_STORAGE_KEY, VOTES_MAX } from './icons.js?v=20260908compareprompts1'; import { showScoreboard } from './scoreboard.js?v=20260909voteconfirmalign1'; diff --git a/static/js/cookbookServe.js b/static/js/cookbookServe.js index 02a189894..5e1699282 100644 --- a/static/js/cookbookServe.js +++ b/static/js/cookbookServe.js @@ -7,7 +7,7 @@ import uiModule from './ui.js?v=20260916largetoolscroll1'; import spinnerModule from './spinner.js'; import { providerLogo } from './providers.js'; -import { modelColor } from './chatRenderer.js?v=20260913richdiff1'; +import { modelColor } from './chatRenderer.js?v=20260914metricssummary1'; import { bindMenuDismiss, dismissOrRemove, diff --git a/static/js/document.js b/static/js/document.js index 56387b240..f7e01702f 100644 --- a/static/js/document.js +++ b/static/js/document.js @@ -7314,7 +7314,7 @@ import { attachColorPicker } from './colorPicker.js?v=20260910eyedropper1'; } } catch (_) {} if (!document.getElementById('notes-pane') && !document.getElementById('notes-pane-backdrop')) return; - import('./notes.js?v=20260910drawmerge1') + import('./notes.js?v=20260911notesselectioncancel1') .then(mod => { const close = mod.closeNotes || mod.closePanel || mod.default?.closeNotes || mod.default?.closePanel; if (typeof close === 'function') close('down'); diff --git a/static/js/emailLibrary.js b/static/js/emailLibrary.js index c1baaa917..385c39ce8 100644 --- a/static/js/emailLibrary.js +++ b/static/js/emailLibrary.js @@ -6,7 +6,7 @@ import spinnerModule from './spinner.js'; import { styledConfirm, showToast, emptyStateIcon } from './ui.js?v=20260916largetoolscroll1'; import { folderDisplayName, sortedFolders } from './emailInbox.js?v=20260914aireply4'; -import settingsModule from './settings.js?v=20260909defaultmodelfix1'; +import settingsModule from './settings.js?v=20260912writingstyle3'; import * as Modals from './modalManager.js'; import { topPortalZ } from './toolWindowZOrder.js'; import { makeWindowDraggable } from './windowDrag.js'; diff --git a/static/js/gallery.js b/static/js/gallery.js index a04fc1378..a8d927951 100644 --- a/static/js/gallery.js +++ b/static/js/gallery.js @@ -2544,7 +2544,7 @@ export function openGallery() { if (visionLink) { visionLink.addEventListener('click', (e) => { e.preventDefault(); - import('./settings.js?v=20260909defaultmodelfix1').then(m => { + import('./settings.js?v=20260912writingstyle3').then(m => { m.open('ai'); // The gallery modal gets a bumped z-index from modalManager; settings // opens with its lower static z-index and lands BEHIND it. Raise it above. diff --git a/static/js/group.js b/static/js/group.js index 6988adc53..39057450f 100644 --- a/static/js/group.js +++ b/static/js/group.js @@ -3,7 +3,7 @@ import uiModule from './ui.js?v=20260916largetoolscroll1'; import markdownModule from './markdown.js'; -import chatRenderer from './chatRenderer.js?v=20260913richdiff1'; +import chatRenderer from './chatRenderer.js?v=20260914metricssummary1'; import spinnerModule from './spinner.js'; import { providerLogo } from './providers.js'; import { PROMPT_TEMPLATES, getUserTemplates } from './presets.js?v=20260908personaname1'; diff --git a/static/js/modelPicker.js b/static/js/modelPicker.js index 2883dd871..e76da0f40 100644 --- a/static/js/modelPicker.js +++ b/static/js/modelPicker.js @@ -3,7 +3,7 @@ import { providerLogo } from './providers.js'; import uiModule from './ui.js?v=20260916largetoolscroll1'; -import settingsModule from './settings.js?v=20260909defaultmodelfix1'; +import settingsModule from './settings.js?v=20260912writingstyle3'; import { sortModelObjects } from './modelSort.js'; import spinnerModule from './spinner.js'; diff --git a/static/js/models.js b/static/js/models.js index b650b0d4b..f932cde0d 100644 --- a/static/js/models.js +++ b/static/js/models.js @@ -9,7 +9,7 @@ import uiModule from './ui.js?v=20260916largetoolscroll1'; import sessionModule from './sessions.js'; import dragSortModule from './dragSort.js'; import spinnerModule from './spinner.js'; -import { modelColor } from './chatRenderer.js?v=20260913richdiff1'; +import { modelColor } from './chatRenderer.js?v=20260914metricssummary1'; import { providerLogo } from './providers.js'; import { sortModelIds } from './modelSort.js'; diff --git a/static/js/sessions.js b/static/js/sessions.js index f28abd4fb..f16b8f934 100644 --- a/static/js/sessions.js +++ b/static/js/sessions.js @@ -3,7 +3,7 @@ import Storage from './storage.js'; import uiModule, { autoResize, styledPrompt } from './ui.js?v=20260916largetoolscroll1'; -import chatRenderer from './chatRenderer.js?v=20260913richdiff1'; +import chatRenderer from './chatRenderer.js?v=20260914metricssummary1'; import { providerLogo } from './providers.js'; import { initModelPicker, updateModelPicker } from './modelPicker.js?v=20260909routeidentity1'; import themeModule from './theme.js?v=20260911organsrain1'; diff --git a/static/js/sidebar-layout.js b/static/js/sidebar-layout.js index 6c0a77b7d..949ec532f 100644 --- a/static/js/sidebar-layout.js +++ b/static/js/sidebar-layout.js @@ -286,6 +286,10 @@ export function initSidebarLayout(Storage, opts) { function checkSidebarAutoCollapse() { if (_userToggledSidebar) return; + // Mobile uses a fixed overlay drawer. Keyboard and orientation changes + // emit resize events there, but should not run the desktop width-based + // auto-collapse logic against an intentionally opened drawer. + if (window.innerWidth < 768) return; const sidebar = document.getElementById('sidebar'); if (!sidebar) return; const isHidden = sidebar.classList.contains('hidden'); @@ -320,7 +324,7 @@ export function initSidebarLayout(Storage, opts) { const isMobileViewport = window.innerWidth < 768; if (_wasMobileViewport && !isMobileViewport) _restoreDesktopSidebarSide(); _wasMobileViewport = isMobileViewport; - _userToggledSidebar = false; // allow auto-collapse on actual resize + if (!isMobileViewport) _userToggledSidebar = false; // allow auto-collapse on desktop resize requestAnimationFrame(checkSidebarAutoCollapse); }); // Re-check when the document split is opened or closed. Do not react to diff --git a/static/js/slashCommands.js b/static/js/slashCommands.js index 2ba1ab8ea..ec4692aa9 100644 --- a/static/js/slashCommands.js +++ b/static/js/slashCommands.js @@ -13,12 +13,12 @@ import Storage from './storage.js'; import uiModule from './ui.js?v=20260916largetoolscroll1'; import sessionModule from './sessions.js'; import modelsModule from './models.js'; -import chatRenderer from './chatRenderer.js?v=20260913richdiff1'; +import chatRenderer from './chatRenderer.js?v=20260914metricssummary1'; import spinnerModule from './spinner.js'; import themeModule from './theme.js?v=20260911organsrain1'; import documentModule from './document.js?v=20260916docctx2'; import workspaceModule from './workspace.js'; -import settingsModule from './settings.js?v=20260909defaultmodelfix1'; +import settingsModule from './settings.js?v=20260912writingstyle3'; import cookbookModule from './cookbook.js'; import { EVAL_PROMPTS } from './compare/index.js?v=20260909mobilepaneaddscroll1'; import { PROVIDER_DEVICE_FLOWS, formatDeviceFlowError, runProviderDeviceFlow } from './providerDeviceFlow.js'; diff --git a/static/js/tasks.js b/static/js/tasks.js index 02b738d2a..5c511dc0c 100644 --- a/static/js/tasks.js +++ b/static/js/tasks.js @@ -800,10 +800,12 @@ function _renderTaskChips() { b.className = 'memory-cat-chip task-filter-chip' + (kind === 'status' ? ' task-status-filter-chip' : '') + (active ? ' active' : ''); b.textContent = label; b.addEventListener('click', () => { - if (kind === 'status') _taskStatusFilter = _taskStatusFilter === value ? null : value; - else { + if (kind === 'status') { + _taskStatusFilter = _taskStatusFilter === value ? null : value; + _taskFilter = null; + } else { _taskFilter = value; - if (value === null) _taskStatusFilter = null; + _taskStatusFilter = null; } _renderList(); }); diff --git a/static/js/theme.js b/static/js/theme.js index 5d552d097..4d87c0248 100644 --- a/static/js/theme.js +++ b/static/js/theme.js @@ -31,6 +31,7 @@ export const THEMES = { arcade: { bg:'#15131d', fg:'#f4e85c', panel:'#0b1720', border:'#305f68', red:'#ff4f91' }, blueprint: { bg:'#10263b', fg:'#e8f1f5', panel:'#091a29', border:'#48758d', red:'#e6b84a' }, monolith: { bg:'#202020', fg:'#d8d8d2', panel:'#121212', border:'#50504b', red:'#b6e05c' }, + yoyo: { bg:'#211f23', fg:'#dfdbd7', panel:'#141415', border:'#3e5146', red:'#e09d6c' }, }; const DEFAULT_THEME = 'dark'; @@ -67,6 +68,7 @@ const THEME_DEFAULT_PATTERN = { arcade: 'synapse', blueprint: 'dots', monolith: 'perlin-flow', + yoyo: 'ascii-fireflies', }; // Default effect colors for specific themes (overrides --fg) @@ -80,6 +82,7 @@ const THEME_DEFAULT_EFFECT_COLOR = { arcade: '#ff4f91', blueprint: '#e6b84a', monolith: '#b6e05c', + yoyo: '#b8e6c1', }; // Default effect intensity (0..1) per theme. Any theme not listed defaults to 1. @@ -87,6 +90,7 @@ const THEME_DEFAULT_INTENSITY = { midnight: 0.5, cyberpunk: 0.55, terminal: 0.8, + yoyo: 0.8, organs: 0.75, }; @@ -694,7 +698,9 @@ export function initThemeUI() { // Render custom theme swatches into separate card const userGrid = document.getElementById('themeUserGrid'); const userCard = document.getElementById('themeUserCard'); - const customEntries = Object.entries(customThemes); + // Hide legacy custom copies after a theme graduates into the preset list. + const customEntries = Object.entries(customThemes) + .filter(([name]) => !THEMES[name]); if (customEntries.length > 0 && userGrid && userCard) { userCard.style.display = ''; userGrid.innerHTML = customEntries.map(([name, c]) => ` diff --git a/static/style.css b/static/style.css index 3c9cf9903..7d0116f47 100644 --- a/static/style.css +++ b/static/style.css @@ -8809,11 +8809,11 @@ pre { background: var(--code-bg, var(--hl-bg, #282c34)) !important; } transform-origin: left center; } .compare-pane.is-streaming { - border-color: color-mix(in srgb, #4da3ff 34%, var(--border)); - box-shadow: inset 0 1px 0 color-mix(in srgb, #4da3ff 12%, transparent); + border-color: color-mix(in srgb, var(--accent, var(--red)) 34%, var(--border)); + box-shadow: inset 0 1px 0 color-mix(in srgb, var(--accent, var(--red)) 12%, transparent); } .compare-pane.is-streaming::before { - background: linear-gradient(90deg, transparent 0%, #4da3ff 35%, color-mix(in srgb, var(--red) 75%, #4da3ff) 50%, #4da3ff 65%, transparent 100%); + background: linear-gradient(90deg, transparent 0%, var(--accent, var(--red)) 35%, var(--red) 50%, var(--accent, var(--red)) 65%, transparent 100%); animation: compare-pane-progress 1.25s linear infinite; } .compare-pane.is-awaiting-input { @@ -9127,9 +9127,9 @@ pre { background: var(--code-bg, var(--hl-bg, #282c34)) !important; } border-color: color-mix(in srgb, var(--border) 70%, transparent); } .pane-mode-search { - color: color-mix(in srgb, #4da3ff 82%, var(--fg)); - background: color-mix(in srgb, #4da3ff 10%, var(--bg)); - border-color: color-mix(in srgb, #4da3ff 28%, var(--border)); + color: color-mix(in srgb, var(--accent, var(--red)) 82%, var(--fg)); + background: color-mix(in srgb, var(--accent, var(--red)) 10%, var(--bg)); + border-color: color-mix(in srgb, var(--accent, var(--red)) 28%, var(--border)); } .pane-mode-research { color: color-mix(in srgb, #36c48f 82%, var(--fg)); @@ -20797,6 +20797,13 @@ body:not(.email-doc-split-active) #email-lib-modal.email-lib-fullscreen:not(.mod flex: 0 0 auto; min-height: 0; } +/* The active Tasks filter row is also a compact toolbar. Keep its wrapper + content-sized so changing a filter cannot make the mobile panel grow a + large empty gap between the chips and the task cards. */ +.admin-card > .doclib-chip-scroll-frame:has(> #tasks-filter-chips) { + flex: 0 0 auto; + min-height: 0; +} #memory-category-filters { flex: 0 0 auto; min-height: 25px; @@ -21227,17 +21234,16 @@ body:not(.email-doc-split-active) #email-lib-modal.email-lib-fullscreen:not(.mod color: color-mix(in srgb, var(--fg) 45%, transparent); } .pdf-loading-state { - min-height: 180px; - margin: 16px; + min-height: 120px; + margin: 0; box-sizing: border-box; flex-direction: column; gap: 10px; - border: 1px solid color-mix(in srgb, var(--accent, var(--red)) 22%, var(--border)); - border-radius: 12px; - background: color-mix(in srgb, var(--accent, var(--red)) 5%, var(--bg)); + border: 0; + border-radius: 0; + background: transparent; color: color-mix(in srgb, var(--fg) 68%, transparent); - box-shadow: inset 0 1px 0 color-mix(in srgb, var(--fg) 7%, transparent), - 0 8px 24px rgba(0, 0, 0, 0.12); + box-shadow: none; font-size: 12px; font-weight: 600; letter-spacing: 0.01em; diff --git a/static/sw.js b/static/sw.js index cbc5d5bf0..8a61c7e40 100644 --- a/static/sw.js +++ b/static/sw.js @@ -62,7 +62,7 @@ const PRECACHE = [ '/static/js/tts-ai.js', '/static/js/document.js?v=20260916docctx2', '/static/js/gallery.js?v=20260910promptcopy1', - '/static/js/chatRenderer.js?v=20260914pdfstrip1', + '/static/js/chatRenderer.js?v=20260914metricssummary1', '/static/js/codeRunner.js?v=20260831richtexttools91', '/static/js/chatStream.js?v=20260914pdfstrip1', '/static/js/chat.js?v=20260917toolttft1', @@ -72,18 +72,18 @@ const PRECACHE = [ '/static/js/compare/vote.js?v=20260828resendcaldrag1', '/static/js/colorPicker.js?v=20260910eyedropper1', '/static/js/panels.js?v=20260909movepicklayer1', - '/static/js/theme.js?v=20260909effectspeed1', + '/static/js/theme.js?v=20260911organsrain1', '/static/js/censor.js', '/static/js/settings.js?v=20260912writingstyle3', '/static/js/admin.js?v=20260914toolschemaprofiles1', '/static/js/init.js?v=20260829chatstyle12', '/static/js/slashCommands.js?v=20260902tuiharness1', '/static/js/research/jobs.js?v=20260910researcherrorpersist1', - '/static/js/emailInbox.js?v=20260903emailsend2', + '/static/js/emailInbox.js?v=20260914aireply4', '/static/js/emailLibrary/utils.js', '/static/js/emailLibrary/signatureFold.js', '/static/js/emailLibrary/state.js', - '/static/js/notes.js?v=20260910drawmerge2', + '/static/js/notes.js?v=20260911notesselectioncancel1', '/static/js/tasks.js?v=20260914taskmodel1', '/static/js/calendar.js?v=20260914emailsource11', '/static/js/calendar/utils.js', diff --git a/tests/test_agent_evidence_loop.py b/tests/test_agent_evidence_loop.py index 6a044a175..fcf6e81b8 100644 --- a/tests/test_agent_evidence_loop.py +++ b/tests/test_agent_evidence_loop.py @@ -15,6 +15,45 @@ def _events(chunks): return parsed +def test_deepseek_flash_visual_continuation_flattens_only_tool_visual_history(): + messages = [ + {"role": "system", "content": "system contract"}, + {"role": "user", "content": "original task"}, + { + "role": "assistant", + "content": None, + "tool_calls": [{"id": "c1", "type": "function", "function": { + "name": "inspect_media", "arguments": "{}", + }}], + }, + {"role": "tool", "tool_call_id": "c1", "content": "preview follows"}, + { + "role": "user", + "metadata": {"source": "tool visual evidence", "trusted": False}, + "content": [ + {"type": "text", "text": "Visual evidence returned by tool execution."}, + {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,AAA"}}, + ], + }, + ] + + flattened = agent_loop._deepseek_flash_visual_continuation(messages, "original task") + + assert [message["role"] for message in flattened] == ["system", "user"] + assert "original task" in flattened[-1]["content"][0]["text"] + assert flattened[-1]["content"][1]["image_url"]["url"].endswith("AAA") + assert agent_loop._deepseek_flash_visual_continuation( + messages[:-1], "original task" + ) is None + + +def test_deepseek_flash_vision_compatibility_is_exact_model_only(): + assert agent_loop._is_deepseek_flash_vision_model("deepseek-flash") + assert agent_loop._is_deepseek_flash_vision_model("provider/deepseek-flash") + assert not agent_loop._is_deepseek_flash_vision_model("deepseek-v4-pro") + assert not agent_loop._is_deepseek_flash_vision_model("deepseek-flash-preview") + + def _patch_loop(monkeypatch, responses, captured_kwargs=None): monkeypatch.setattr(agent_loop, "get_setting", lambda key, default=None: default) monkeypatch.setattr(agent_loop, "get_mcp_manager", lambda: None) @@ -796,3 +835,13 @@ def test_run_the_script_is_not_inferred_when_multiple_scripts_are_named(): ) assert agent_loop._requested_verification_command(request) == "" + + +def test_inspect_saved_file_requests_artifact_readback_verification(): + request = ( + "Create /workspace/output/results.csv, inspect the saved file, " + "then summarize completion." + ) + + assert agent_loop._requested_post_edit_verification(request) + assert agent_loop._requested_artifact_readback(request) diff --git a/tests/test_chat_ttft_timer_static.py b/tests/test_chat_ttft_timer_static.py index a47ca2018..d5d28b430 100644 --- a/tests/test_chat_ttft_timer_static.py +++ b/tests/test_chat_ttft_timer_static.py @@ -27,8 +27,10 @@ def test_compact_footer_and_details_show_real_performance_counters(): assert "`${Number(tps).toFixed(2)} tok/s`" in RENDERER assert "`${Number(ttft).toFixed(3)}s TTFT`" in RENDERER assert "`${Number(injectedTokens).toLocaleString()} in`" in RENDERER - assert 'Input (all rounds)' in RENDERER - assert 'Injected (first request)' in RENDERER + assert 'Input' in RENDERER + assert 'Injected' in RENDERER + assert 'all rounds' not in RENDERER + assert 'first request' not in RENDERER assert 'Tool schemas' in RENDERER assert 'Agent rounds' in RENDERER assert 'Tool calls' in RENDERER diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 264aab082..5ac2ae625 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -382,13 +382,36 @@ def test_native_execution_limits_allow_multi_artifact_work_without_unbounded_rou def test_compact_preview_honors_configured_interactive_round_limit(): - assert interactive_execution_limit(100) == 8 - assert interactive_execution_limit(1000) == 8 + # A configured budget is the user's explicit instruction, honored up to + # the same 200 ceiling the settings endpoint enforces. + assert interactive_execution_limit(100) == 100 + assert interactive_execution_limit(1000) == 200 assert interactive_execution_limit(0) == 1 + # No resolvable budget falls back to the bounded default. assert interactive_execution_limit(None) == 8 assert interactive_execution_limit("invalid") == 8 +def test_compact_preview_honors_configured_tool_call_budget(): + from src.clean_agent_preview import ( + INTERACTIVE_BROWSER_TOOL_CALL_LIMIT, + INTERACTIVE_TOOL_CALL_LIMIT, + UNLIMITED_TOOL_CALL_LIMIT, + interactive_tool_call_limit, + ) + # 0 means unlimited, matching the main agent loop's max_tool_calls <= 0. + assert interactive_tool_call_limit(0) == UNLIMITED_TOOL_CALL_LIMIT + assert interactive_tool_call_limit(0, browser_offered=True) == UNLIMITED_TOOL_CALL_LIMIT + # An explicit finite budget is honored in both directions. + assert interactive_tool_call_limit(5) == 5 + assert interactive_tool_call_limit(250) == 250 + # A malformed value keeps the bounded default. + assert interactive_tool_call_limit("invalid") == INTERACTIVE_TOOL_CALL_LIMIT + assert interactive_tool_call_limit(None, browser_offered=True) == ( + INTERACTIVE_BROWSER_TOOL_CALL_LIMIT + ) + + @pytest.mark.parametrize('text,expected', [ ('helo', True), ('Hello!', True), ('thanks', True), ('hello, find the latest news', False), ('thanks, now open the source', False), @@ -5256,7 +5279,9 @@ async def test_model_choice_budget_preserves_evidence_for_final_answer_without_e import src.clean_agent_preview as module # Exercise the boundary deterministically without coupling this recovery # test to the larger production allowance for interactive turns. - monkeypatch.setattr(module, 'INTERACTIVE_TOOL_CALL_LIMIT', 6) + # The budget is the configured agent_max_tool_calls setting; the module + # constant is only the fallback when no budget is resolvable. + _tool_budget = 6 def call(i): return {'index': i, 'id': f'call-{i}', 'function': { 'name': 'web_fetch', 'arguments': json.dumps({'url': f'https://example.org/{i}'})}} @@ -5291,7 +5316,8 @@ async def test_model_choice_budget_preserves_evidence_for_final_answer_without_e routing_experiment='recent_model_choice') raw = [chunk async for chunk in stream_preview( endpoint_url='http://test', model='test', messages=[{'role':'user','content':'Read these public pages.'}], - headers={}, turn_contract=contract, session_id='test', owner='test', disabled_tools=set(), tool_policy=ToolPolicy())] + headers={}, turn_contract=contract, session_id='test', owner='test', disabled_tools=set(), + tool_policy=ToolPolicy(), max_tool_calls=_tool_budget)] assert len(executions) == 6 assert len(requests) == 3 assert 'tools' not in requests[-1] @@ -5898,9 +5924,9 @@ async def test_context_overflow_retries_model_request_without_replaying_tool(mon import src.clean_agent_preview as module requests, executions = [], [] overflow_request = 2 if recovery_kind == 'none' else 3 + _tool_budget = 0 if recovery_kind == 'budget': - monkeypatch.setattr(module, 'INTERACTIVE_TOOL_CALL_LIMIT', 1) - monkeypatch.setattr(module, 'INTERACTIVE_BROWSER_TOOL_CALL_LIMIT', 1) + _tool_budget = 1 async def handle(request): payload = json.loads(request.content) requests.append(payload) @@ -5936,7 +5962,7 @@ async def test_context_overflow_retries_model_request_without_replaying_tool(mon raw = [chunk async for chunk in stream_preview( endpoint_url='http://test', model='test', headers={}, turn_contract=contract, messages=[{'role': 'user', 'content': prompt}], session_id='test', owner='test', - disabled_tools=set(), tool_policy=ToolPolicy())] + disabled_tools=set(), tool_policy=ToolPolicy(), max_tool_calls=_tool_budget)] assert any('The page was read.' in chunk for chunk in raw) assert len(executions) == 1 assert len(requests) == overflow_request + 1 @@ -7075,3 +7101,80 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch): ] assert "Final completion round" in messages[-1]["content"] assert messages[-2]["content"][1]["type"] == "image_url" +@pytest.mark.asyncio +async def test_tool_round_leadin_is_replaced_by_terminal_synthesis(monkeypatch): + from dataclasses import replace + import src.clean_agent_preview as module + + packets = iter([ + {'choices': [{'delta': { + 'content': "I'll browse IKEA and look at their chairs.", + 'tool_calls': [{'index': 0, 'id': 'browse-1', 'function': { + 'name': 'private_browser', + 'arguments': json.dumps({ + 'action': 'open', + 'url': 'https://www.ikea.com/us/en/cat/armchairs-16239/', + }), + }}], + }}]}, + {'choices': [{'delta': { + 'reasoning_content': 'I have enough. Now compose the final answer.', + 'content': 'The most epic option is the DYVLINGE swivel chair.', + }}]}, + ]) + + class Response: + def __init__(self, payload): self.payload = payload + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def raise_for_status(self): pass + async def aiter_lines(self): + yield 'data: ' + json.dumps(self.payload) + yield 'data: [DONE]' + + class Client: + def __init__(self, **kwargs): pass + async def __aenter__(self): return self + async def __aexit__(self, *args): pass + def stream(self, *args, **kwargs): return Response(next(packets)) + + async def execute(block, **kwargs): + return 'private_browser', {'output': 'IKEA chair evidence', 'exit_code': 0} + + monkeypatch.setattr(module.httpx, 'AsyncClient', Client) + monkeypatch.setattr(module, 'execute_tool_block', execute) + schema = next( + item for item in FUNCTION_TOOL_SCHEMAS + if item['function']['name'] == 'private_browser' + ) + contract = replace( + resolve_full_inventory_contract(schemas=[schema], policy=ToolPolicy()), + routing_experiment='recent_model_choice', + ) + raw = [chunk async for chunk in stream_preview( + endpoint_url='http://test', model='deepseek-test', headers={}, + turn_contract=contract, + messages=[{'role': 'user', 'content': 'Go to ikea.com and find the most epic chair.'}], + session_id='test', owner='test', disabled_tools=set(), + tool_policy=ToolPolicy(), max_rounds=3, + )] + + events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk] + leadin = next(event for event in events if event.get('delta', '').startswith("I'll browse")) + assert 'replacement_scope' not in leadin + synthesis = next( + event for event in events + if event.get('delta', '').startswith('The most epic option') + ) + assert synthesis['render_owner'] == 'streamed' + assert synthesis['replacement_scope'] == 'turn' + metrics = next(event['data'] for event in events if event.get('type') == 'metrics') + assistant_rounds = [ + item.get('content') + for item in metrics['clean_v3_turn'] + if item.get('role') == 'assistant' and item.get('content') + ] + assert assistant_rounds == [ + "I'll browse IKEA and look at their chairs.", + 'The most epic option is the DYVLINGE swivel chair.', + ] diff --git a/tests/test_embeddings_thread_limit.py b/tests/test_embeddings_thread_limit.py new file mode 100644 index 000000000..5225d4b43 --- /dev/null +++ b/tests/test_embeddings_thread_limit.py @@ -0,0 +1,51 @@ +import sys +import types + +import pytest + +from src import embeddings + + +def _install_fastembed(monkeypatch): + calls = [] + module = types.ModuleType("fastembed") + + class TextEmbedding: + def __init__(self, **kwargs): + calls.append(kwargs) + + module.TextEmbedding = TextEmbedding + monkeypatch.setitem(sys.modules, "fastembed", module) + return calls + + +def test_fastembed_threads_default_is_unchanged(monkeypatch, tmp_path): + calls = _install_fastembed(monkeypatch) + monkeypatch.delenv("FASTEMBED_THREADS", raising=False) + monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path)) + + embeddings.FastEmbedClient(model="test-model") + + assert calls == [{"model_name": "test-model", "cache_dir": str(tmp_path)}] + + +def test_fastembed_threads_can_be_bounded(monkeypatch, tmp_path): + calls = _install_fastembed(monkeypatch) + monkeypatch.setenv("FASTEMBED_THREADS", "2") + monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path)) + + embeddings.FastEmbedClient(model="test-model") + + assert calls == [ + {"model_name": "test-model", "cache_dir": str(tmp_path), "threads": 2} + ] + + +@pytest.mark.parametrize("value", ["0", "257", "not-a-number"]) +def test_fastembed_threads_rejects_invalid_values(monkeypatch, tmp_path, value): + _install_fastembed(monkeypatch) + monkeypatch.setenv("FASTEMBED_THREADS", value) + monkeypatch.setattr(embeddings, "FASTEMBED_CACHE_DIR", str(tmp_path)) + + with pytest.raises(ValueError, match="FASTEMBED_THREADS"): + embeddings.FastEmbedClient(model="test-model") diff --git a/tests/test_lmstudio_vision.py b/tests/test_lmstudio_vision.py index a4ed78e2b..1add5f128 100644 --- a/tests/test_lmstudio_vision.py +++ b/tests/test_lmstudio_vision.py @@ -95,6 +95,8 @@ class TestModelSupportsVision: def test_falls_back_to_name_when_no_endpoint(self): # No endpoint URL → pure name heuristic. assert chat_helpers.model_supports_vision("llava-1.6", "") is True + assert chat_helpers.model_supports_vision("deepseek-flash", "") is True + assert chat_helpers.model_supports_vision("deepseek-v4-pro", "") is False assert chat_helpers.model_supports_vision("mistral-7b", "") is False def test_falls_back_to_name_when_endpoint_unknown(self, monkeypatch): diff --git a/tests/test_review_20260919_fixes.py b/tests/test_review_20260919_fixes.py new file mode 100644 index 000000000..4fb1ce179 --- /dev/null +++ b/tests/test_review_20260919_fixes.py @@ -0,0 +1,108 @@ +"""Regressions for the 2026-09-19 review fixes. + +Each test fails on the pre-fix tree; see ~/odysseus-review-20260919.md. +""" +import json + +import pytest + +from services.search import core as search_core +from src.agent_loop import _read_file_block_path, _read_file_targets_artifact +from src.turn_contract import personal_store_families, requested_capabilities + + +@pytest.mark.parametrize("message,family", [ + ("is there anything new in my inbox?", "email"), + ("check my email for news from Bob", "email"), + ("show me my latest email updates", "email"), + ("give me an update on my tasks", "tasks"), + ("what's new on my calendar this week?", "calendar"), + ("catch me up on my meetings", "calendar"), + ("anything new in my documents?", "documents"), + ("what should i know about my notes", "notes"), +]) +def test_briefing_phrase_does_not_hijack_personal_store(message, family): + """A broad-briefing phrase must not route the user's own store to the Web.""" + capabilities = requested_capabilities(message) + assert family in capabilities + assert "search_browser" not in capabilities + + +@pytest.mark.parametrize("message", [ + "what's new in AI this week", + "give me an update on the war in Ukraine", + "what are the latest events in Kyiv", + "news about the fed rate decision", + "what should i know about rust 2.0", + "catch me up on the latest AI news", + "latest info on the iphone 18", +]) +def test_open_web_briefings_keep_their_web_route(message): + """Open-web subjects that borrow a product noun keep search_browser.""" + assert requested_capabilities(message) == frozenset({"search_browser"}) + + +def test_personal_store_families_requires_possessive(): + assert personal_store_families("my calendar") == frozenset({"calendar"}) + assert personal_store_families("the latest events in Kyiv") == frozenset() + + +NGINX = { + "title": "NGINX Documentation", + "snippet": "Official configuration docs for NGINX", + "url": "https://nginx.org/en/docs/", +} +IKEA = { + "title": "BILLY Bookcase Assembly Manual", + "snippet": "IKEA BILLY instructions PDF", + "url": "https://ikea.com/manuals/billy", +} +FORD = { + "title": "Ford Ranger Owner Manual", + "snippet": "Operator manual for the Ford Ranger", + "url": "https://ford.com/manual", +} + + +@pytest.mark.parametrize("query,result", [ + ("how to configure nginx docs", NGINX), + ("where can i find the ikea billy manual", IKEA), + ("best guide for sourdough", {"title": "Sourdough Starter Guide", + "snippet": "A complete guide to sourdough", + "url": "https://ex.com/sourdough"}), + ("nginx configuration docs", NGINX), +]) +def test_document_lookups_accept_relevant_results(query, result): + """Leading function words and task verbs are not the query entity.""" + assert search_core._result_has_query_overlap(query, result) is True + + +@pytest.mark.parametrize("query", [ + "ikea billy manual", + "nginx docs", + "sourdough guide", +]) +def test_document_lookups_still_reject_homonyms(query): + """A result naming none of the entity terms is still not evidence.""" + assert search_core._result_has_query_overlap(query, FORD) is False + + +def test_read_file_block_path_accepts_json_and_bare_text(): + assert _read_file_block_path( + json.dumps({"path": "/workspace/out/report.md"}) + ) == "/workspace/out/report.md" + assert _read_file_block_path( + "/workspace/out/report.md\ntrailing" + ) == "/workspace/out/report.md" + assert _read_file_block_path("") == "" + + +def test_reading_the_input_does_not_verify_the_output(): + """The forced read-back must target the deliverable, not the source.""" + target = "/workspace/out/report.md" + assert _read_file_targets_artifact(json.dumps({"path": target}), target) + assert _read_file_targets_artifact(json.dumps({"path": "./report.md"}), target) + assert not _read_file_targets_artifact( + json.dumps({"path": "/workspace/input/data.csv"}), target + ) + assert not _read_file_targets_artifact(json.dumps({"path": target}), None) diff --git a/tests/test_service_search_provider_guards.py b/tests/test_service_search_provider_guards.py index 1277370a9..29b1b36c4 100644 --- a/tests/test_service_search_provider_guards.py +++ b/tests/test_service_search_provider_guards.py @@ -752,3 +752,21 @@ def test_service_ddg_html_fallback_sends_safesearch(monkeypatch): assert seen["params"]["kp"] == "-2" assert seen["timeout"] <= 5 assert results[0]["url"].startswith("https://notduckduckgo.com/") +def test_relevance_filter_falls_back_when_heuristic_rejects_every_result(): + rows = [{ + "title": "Bicycle buying guide", + "snippet": "Road bikes and city bikes", + "url": "https://example.test/bicycle-guide", + }] + + assert core._filter_low_relevance_results("bicycles", rows) == rows + + +def test_relevance_filter_does_not_bypass_explicit_site_scope(): + rows = [{ + "title": "Bicycle buying guide", + "snippet": "Road bikes and city bikes", + "url": "https://example.test/bicycle-guide", + }] + + assert core._filter_low_relevance_results("site:official.test bicycles", rows) == [] diff --git a/tests/test_tool_path_confinement.py b/tests/test_tool_path_confinement.py index f9fe4fc81..be4a75162 100644 --- a/tests/test_tool_path_confinement.py +++ b/tests/test_tool_path_confinement.py @@ -338,3 +338,8 @@ async def test_write_file_dispatch_blocks_cron(monkeypatch): ) assert "outside the allowed roots" in (result.get("error") or "") assert result.get("exit_code") == 1 +@pytest.mark.parametrize("filename", ["auth.json", "app.db", "settings.json"]) +def test_application_secrets_are_sensitive_paths(filename): + from src.tool_execution import _is_sensitive_path + + assert _is_sensitive_path(f"/tmp/odysseus-data/{filename}") diff --git a/tests/test_tool_routing_experiment.py b/tests/test_tool_routing_experiment.py index 0aacbea78..d09caf00e 100644 --- a/tests/test_tool_routing_experiment.py +++ b/tests/test_tool_routing_experiment.py @@ -21,6 +21,46 @@ def test_disabled_ocr_is_not_reintroduced_by_explicit_request_or_warm_history(): assert 'extract_text' not in offered.executable +def test_web_subject_followup_does_not_offer_shell_or_private_stores(): + from src.tool_routing_experiment import MODEL_CHOICE_MODE + from src.turn_contract import requested_capabilities + + policy = ToolPolicy() + history = [ + {'role': 'user', 'content': 'Is there Rocket League for Switch 2?'}, + {'role': 'assistant', 'content': 'It uses the Switch listing.', 'metadata': { + 'tool_events': [ + {'tool': 'web_search', 'exit_code': 0, 'error': False}, + {'tool': 'web_fetch', 'exit_code': 0, 'error': False}, + ], + }}, + ] + capabilities = requested_capabilities( + 'I searched rocket and cannot find it', history, + ) + inventory = resolve_full_inventory_contract( + schemas=FUNCTION_TOOL_SCHEMAS, policy=policy, + ) + routed = resolve_turn_contract( + capabilities=capabilities, schemas=FUNCTION_TOOL_SCHEMAS, policy=policy, + ) + result = select_experiment_inventory( + inventory, routed, history, MODEL_CHOICE_MODE, + user_text='I searched rocket and cannot find it', + ) + + assert capabilities == {'search_browser'} + assert result.offered + assert {schema.rsplit('__', 1)[-1] for schema in result.offered} <= { + 'web_search', 'web_fetch', 'private_browser', 'youtube_tool', + 'search_hf_models', 'pdf_extract', + } + assert not result.offered.intersection({ + 'bash', 'python', 'read_file', 'manage_notes', 'manage_documents', + 'search_emails', 'search_chats', + }) + + def test_model_choice_rollout_is_account_model_and_header_scoped(): from src.tool_routing_experiment import MODEL_CHOICE_MODE, MODEL_CHOICE_MODEL assert experiment_mode(None, 'pewds', MODEL_CHOICE_MODEL) == MODEL_CHOICE_MODE diff --git a/tests/test_turn_contract.py b/tests/test_turn_contract.py index 93def82d6..f2755538b 100644 --- a/tests/test_turn_contract.py +++ b/tests/test_turn_contract.py @@ -12,13 +12,59 @@ from src.tool_policy import ToolPolicy, WEB_ACCESS_TOOL_NAMES from src.tool_schemas import FUNCTION_TOOL_SCHEMAS from src.turn_contract import ( FAMILY_TOOLS, RequiredReadOperation, active_turn_contract, bind_turn_contract, canonical_tool, - requested_capabilities, required_read_operation_for_request, + immediately_established_family, requested_capabilities, required_read_operation_for_request, requests_independent_web_source, requests_supporting_web_source, resolve_turn_contract, preserve_bound_editor_selected_tools, selected_tools_for_request, targets_bound_editor_request, ) +def _completed_tool_turn(user_text, *tools): + return [ + {"role": "user", "content": user_text}, + {"role": "assistant", "content": "Done", "metadata": {"tool_events": [ + {"tool": tool, "exit_code": 0, "error": False} for tool in tools + ]}}, + ] + + +@pytest.mark.parametrize(("prior", "followup"), [ + ("Is there Rocket League for Switch 2?", "I searched rocket and cannot find it"), + ("Find the current price of the Framework laptop", "framework is not showing for me"), + ("Look up the Kyoto railway museum opening hours", "I cannot find the kyoto museum result"), + ("Check whether Aurora 7 is available on PlayStation", "where is aurora 7 listed"), +]) +def test_subject_continuity_keeps_public_web_followups_out_of_shell(prior, followup): + history = _completed_tool_turn(prior, "web_search", "web_fetch") + + assert immediately_established_family(followup, history) == "search_browser" + assert requested_capabilities(followup, history) == frozenset({"search_browser"}) + + +def test_subject_continuity_uses_typed_private_domain_without_crossing_to_shell(): + history = _completed_tool_turn("Find my Project Juniper note", "manage_notes") + + assert requested_capabilities( + "juniper is not showing in the results", history, + ) == frozenset({"notes"}) + + +def test_explicit_domain_switch_overrides_subject_continuity(): + history = _completed_tool_turn("Find the current Orion browser release", "web_search") + + assert requested_capabilities( + "search my documents for Orion", history, + ) == frozenset({"documents"}) + + +def test_mixed_domain_turn_is_not_inherited_as_one_domain(): + history = _completed_tool_turn( + "Find sources for Atlas and save them to my notes", "web_search", "manage_notes", + ) + + assert immediately_established_family("atlas is missing", history) is None + + @pytest.mark.parametrize('prompt', [ "What's new in Sweden?", "What's happening in Sweden?", diff --git a/tests/test_yoyo_builtin_theme.py b/tests/test_yoyo_builtin_theme.py new file mode 100644 index 000000000..3845c099e --- /dev/null +++ b/tests/test_yoyo_builtin_theme.py @@ -0,0 +1,21 @@ +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] + + +def test_yoyo_is_a_builtin_theme_with_its_saved_effect_defaults(): + source = (ROOT / "static/js/theme.js").read_text(encoding="utf-8") + + assert "yoyo: { bg:'#211f23', fg:'#dfdbd7'" in source + assert "yoyo: 'ascii-fireflies'" in source + assert "yoyo: '#b8e6c1'" in source + assert ".filter(([name]) => !THEMES[name])" in source + + +def test_agent_theme_inventory_includes_yoyo(): + interaction = (ROOT / "src/ai_interaction.py").read_text(encoding="utf-8") + schemas = (ROOT / "src/tool_schemas.py").read_text(encoding="utf-8") + + assert '"monolith", "yoyo"' in interaction + assert "blueprint, monolith, yoyo" in schemas