Consolidate Odysseus agent harness and tool contracts

This commit is contained in:
pewdiepie-archdaemon
2026-09-17 10:07:40 +00:00
parent 84aa9a91de
commit 218d762427
229 changed files with 28899 additions and 1551 deletions
+2 -1
View File
@@ -560,7 +560,8 @@ async def do_manage_settings(content: str, owner: Optional[str] = None) -> Dict:
"hard max": "agent_input_token_hard_max",
"token budget cap": "agent_input_token_hard_max",
"input budget cap": "agent_input_token_hard_max",
"writing style": "email_writing_style", "email writing style": "email_writing_style",
"writing style": "document_writing_style", "document writing style": "document_writing_style",
"email writing style": "email_writing_style",
"reply writing style": "email_writing_style", "email reply writing style": "email_writing_style",
}
def _resolve(k):
+56 -2
View File
@@ -11,6 +11,38 @@ from src.upload_handler import reserve_upload_references
logger = logging.getLogger(__name__)
_DOCUMENT_SEARCH_STOPWORDS = frozenset({
'a', 'an', 'and', 'any', 'about', 'document', 'documents', 'for', 'in',
'my', 'of', 'on', 'or', 'plans', 'the', 'to',
})
def _document_search_tokens(value: str) -> list[str]:
return [
token for token in re.findall(r'[a-z0-9]+', str(value or '').lower())
if token not in _DOCUMENT_SEARCH_STOPWORDS
]
def _rank_document_search(docs, search_text: str):
"""Prefer phrase/all-term matches, then broaden to any meaningful term."""
query = str(search_text or '').strip().lower()
terms = _document_search_tokens(query)
scored = []
for position, doc in enumerate(docs):
haystack = ' '.join((
str(getattr(doc, 'title', '') or ''),
str(getattr(doc, 'current_content', '') or ''),
)).lower()
haystack_terms = set(_document_search_tokens(haystack))
matched = sum(term in haystack_terms for term in terms)
strict = bool(query and query in haystack) or bool(terms and matched == len(terms))
scored.append((doc, strict, matched, position))
strict_matches = [row for row in scored if row[1]]
candidates = strict_matches or [row for row in scored if row[2] > 0]
return [row[0] for row in sorted(candidates, key=lambda row: (-row[2], row[3]))]
def _missing_document_upload(owner: Optional[str], content: Any) -> Optional[str]:
"""Reserve explicit upload URLs before an agent persists document text."""
return reserve_upload_references(get_upload_handler(), owner, content)
@@ -629,6 +661,12 @@ class UpdateDocumentTool:
if is_email_doc:
doc.language = "email"
if new_content == (doc.current_content or ""):
return {
"error": "No update applied — replacement content is unchanged",
"exit_code": 1,
}
missing_id = _missing_document_upload(owner, new_content)
if missing_id:
return {
@@ -761,6 +799,10 @@ class EditDocumentTool:
skipped = 0
for edit in edits:
_find = edit["find"]
if _find == edit["replace"]:
logger.warning("edit_document: skipping no-op FIND/REPLACE block")
skipped += 1
continue
if _find in updated_content:
updated_content = updated_content.replace(_find, edit["replace"], 1)
applied += 1
@@ -941,10 +983,22 @@ class ManageDocumentTool:
search_text = re.sub(
r"\s+(?:instead|please)\s*$", "", search_text, flags=re.IGNORECASE
).strip()
q = q.filter(Document.title.ilike(f"%{search_text}%"))
if args.get("language"):
q = q.filter(Document.language == args["language"])
docs = q.order_by(Document.updated_at.desc()).limit(args.get("limit", 50)).all()
requested_limit = args.get("limit", 50)
try:
requested_limit = max(1, min(int(requested_limit), 200))
except (TypeError, ValueError):
requested_limit = 50
q = q.order_by(Document.updated_at.desc())
# A plain listing must not load the entire document library
# (including every document body) before applying its limit.
if not search_text:
q = q.limit(requested_limit)
docs = q.all()
if search_text:
docs = _rank_document_search(docs, search_text)
docs = docs[:requested_limit]
if not docs:
msg = "No documents found" + (f" matching '{search_text}'" if search_text else "") + "."
return {"response": msg, "documents": [], "exit_code": 0}
+18
View File
@@ -20,6 +20,10 @@ _CODENAV_MAX_LINE = 400
_STRUCTURED_DOCUMENT_SUFFIXES = frozenset({
".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx",
})
_BINARY_ARTIFACT_SUFFIXES = _STRUCTURED_DOCUMENT_SUFFIXES | frozenset({
".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".mp3", ".mp4", ".ogg",
".png", ".wav", ".webm", ".webp", ".zip",
})
def _glob_to_regex(pat: str) -> "re.Pattern":
@@ -275,6 +279,20 @@ class WriteFileTool:
),
"exit_code": 1,
}
# write_file is a UTF-8 text writer. Refuse to silently destroy an
# existing PDF, image, archive, or media artifact produced by a
# format-aware tool, especially after the agent has verified it.
suffix = os.path.splitext(path)[1].casefold()
if suffix in _BINARY_ARTIFACT_SUFFIXES:
target_existed = os.path.isfile(path)
return {
"error": (
f"write_file: refusing UTF-8 text for binary artifact path {path}. "
"Use Python or a format-specific creation tool, then inspect the result."
),
"exit_code": 1,
"binary_artifact_preserved": target_existed,
}
try:
def _write():
old = ""
+72 -5
View File
@@ -433,6 +433,12 @@ class ExtractTextTool:
return {"error": "extract_text unknown argument(s): " + ", ".join(unknown), "exit_code": 1}
try:
raw_path = str(args.get("path") or '')
# Some native-schema models serialize a workspace path using the
# same URI shape as uploads. This alias grants no extra access:
# convert it back to /workspace and let the normal confinement
# resolver enforce the active root.
if raw_path.startswith('odysseus://workspace/'):
raw_path = '/workspace/' + raw_path[len('odysseus://workspace/'):]
if raw_path.startswith('odysseus://'):
# Upload access is independent of a filesystem workspace and
# must never inherit an administrator's cross-owner override.
@@ -450,8 +456,9 @@ class ExtractTextTool:
path = _resolve_media_path(raw_path, tool_name="extract_text")
except ValueError as exc:
return {"error": str(exc), "exit_code": 1}
if path.suffix.casefold() not in _IMAGE_SUFFIXES:
return {"error": "extract_text currently supports local image files", "exit_code": 1}
suffix = path.suffix.casefold()
if suffix not in _IMAGE_SUFFIXES | _PDF_SUFFIXES:
return {"error": "extract_text supports local image and PDF files", "exit_code": 1}
mode = str(args.get("mode") or "all").strip().casefold()
try:
minimum, maximum = float(args.get("min_confidence", .5)), int(args.get("max_results", 512))
@@ -461,7 +468,56 @@ class ExtractTextTool:
return {"error": "invalid extract_text mode or bounds", "exit_code": 1}
try:
from .ocr_engine import extract_image_text
evidence = await asyncio.to_thread(extract_image_text, path, include_layout=bool(args.get("include_layout", False)), numeric_only=mode == "numbers", min_confidence=minimum, max_results=maximum)
if suffix in _PDF_SUFFIXES:
def _extract_pdf_pages():
try:
import pypdfium2 as pdfium
except ImportError as exc:
raise RuntimeError(
"PDF OCR requires the optional pypdfium2 package"
) from exc
document = pdfium.PdfDocument(str(path))
page_count = len(document)
lines, accepted = [], 0
# Keep one OCR call bounded while covering ordinary
# documents completely. Larger PDFs can be inspected in
# page ranges with inspect_media.
rendered_count = min(page_count, 12)
with tempfile.TemporaryDirectory(prefix="odysseus-pdf-ocr-") as temp_dir:
for index in range(rendered_count):
rendered = document[index].render(scale=2.0).to_pil().convert("RGB")
image_path = Path(temp_dir) / f"page-{index + 1}.png"
rendered.save(image_path, "PNG")
remaining = max(1, maximum - len(lines))
page_evidence = extract_image_text(
image_path,
include_layout=bool(args.get("include_layout", False)),
numeric_only=mode == "numbers",
min_confidence=minimum,
max_results=remaining,
)
accepted += int(page_evidence.get("count") or 0)
for line in page_evidence.get("lines") or []:
if len(lines) >= maximum:
break
lines.append({"page": index + 1, **line})
return {
"legend": {
"page": "one-based PDF page",
"t": "text",
"p": "confidence",
"xy": "pixel center",
},
"page_count": page_count,
"pages_processed": rendered_count,
"count": accepted,
"returned": len(lines),
"truncated": accepted > len(lines) or page_count > rendered_count,
"lines": lines,
}
evidence = await asyncio.to_thread(_extract_pdf_pages)
else:
evidence = await asyncio.to_thread(extract_image_text, path, include_layout=bool(args.get("include_layout", False)), numeric_only=mode == "numbers", min_confidence=minimum, max_results=maximum)
except Exception as exc:
return {"error": f"extract_text failed: {exc}", "exit_code": 1}
return {"output": json.dumps(evidence, ensure_ascii=False, separators=(",", ":")), "exit_code": 0, "ocr": evidence}
@@ -715,9 +771,16 @@ class InspectMediaTool:
suffix = path.suffix.lower()
if suffix in _SVG_SUFFIXES:
renderer = shutil.which("rsvg-convert")
renderer_kind = "rsvg"
if not renderer:
renderer = shutil.which("convert")
renderer_kind = "imagemagick"
if not renderer:
return {
"error": "inspect_media SVG rendering requires rsvg-convert",
"error": (
"inspect_media SVG rendering requires rsvg-convert "
"or ImageMagick convert"
),
"exit_code": 1,
}
raw_output = str(args.get("output_path") or "").strip()
@@ -740,7 +803,11 @@ class InspectMediaTool:
output = Path(temporary.name)
rendered = await asyncio.to_thread(
_run,
[renderer, "--output", str(output), str(path)],
(
[renderer, "--output", str(output), str(path)]
if renderer_kind == "rsvg"
else [renderer, str(path), str(output)]
),
60,
)
if rendered.returncode != 0 or not output.is_file() or output.stat().st_size == 0:
@@ -133,6 +133,62 @@ async def list_models(content: str, session_id: Optional[str] = None, owner: Opt
keyword = content.strip().lower() if content.strip() else None
# ``list_models`` historically treated every filter as a literal model-ID
# substring. For recommendation terms that produced an empty catalog even
# though Odysseus already has a hardware detector and fit ranker. Preserve
# the catalog behavior for real model/provider filters, but give these
# semantic filters their expected read-only meaning.
if keyword in {
"recommended", "recommendation", "recommendations",
"compatible", "hardware", "hardware fit", "best fit",
}:
from src.tools.system import do_app_api
fit_result = await do_app_api(json.dumps({
"action": "call",
"method": "GET",
"path": "/api/hwfit/models",
"query": {"fit_only": "true", "limit": 5, "sort": "fit"},
}), owner=owner)
payload = fit_result.get("json") if isinstance(fit_result, dict) else None
system = payload.get("system") if isinstance(payload, dict) else None
models = payload.get("models") if isinstance(payload, dict) else None
if isinstance(system, dict) and isinstance(models, list):
gpu = system.get("gpu_name") or "No GPU detected"
vram = system.get("gpu_vram_gb")
count = system.get("gpu_count")
backend = system.get("backend") or "unknown"
lines = [
"Detected hardware:",
f"- GPU: {gpu}; count={count}; total VRAM={vram} GB; backend={backend}",
f"- CPU: {system.get('cpu_name') or 'unknown'}; RAM={system.get('total_ram_gb')} GB",
"Ranked compatible models:",
]
compact_models = []
for model_row in models[:5]:
if not isinstance(model_row, dict):
continue
compact = {
key: model_row.get(key)
for key in (
"name", "parameter_count", "quant", "required_gb",
"fit_level", "run_mode", "speed_tps", "score", "context",
)
}
compact_models.append(compact)
lines.append(
"- {name}: params={parameter_count}, quant={quant}, required={required_gb} GB, "
"fit={fit_level}, mode={run_mode}, speed={speed_tps} tok/s, score={score}, context={context}".format(
**compact
)
)
return {
"output": "\n".join(lines),
"system": system,
"models": compact_models,
"exit_code": 0,
}
return fit_result
db = SessionLocal()
try:
query = db.query(ModelEndpoint).filter(ModelEndpoint.is_enabled == True)
+17 -3
View File
@@ -874,6 +874,15 @@ def _python_with_visible_final_expression(content: str) -> str:
return ast.unparse(tree)
def _python_with_configured_import_paths(content: str, env: dict | None) -> str:
"""Expose only explicitly configured package roots under Python ``-I``."""
raw = str((env or {}).get("ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES", ""))
paths = [item for item in raw.split(os.pathsep) if item and os.path.isabs(item)]
if not paths:
return content
return f"import site\n[site.addsitedir(path) for path in {paths!r}]\nexec(compile({content!r}, '<odysseus-python-tool>', 'exec'))"
class PythonTool:
async def execute(self, content: str, ctx: dict) -> dict:
from src.tool_execution import agent_cwd, _truncate
@@ -917,7 +926,9 @@ class PythonTool:
# process-global `/workspace` symlink would break concurrent tasks.
# Give Python the same per-task namespace Bash receives so both inline
# code and loaded scripts see the stable virtual workspace root.
namespaced_content = _python_with_visible_final_expression(content)
namespaced_content = _python_with_configured_import_paths(
_python_with_visible_final_expression(content), _subproc_env
)
python_command = shlex.join((sys.executable or "python", "-I", "-c", namespaced_content))
# Code that explicitly uses the public /workspace path runs inside a
# namespace whose stable cwd is that same bind. Host workspaces under
@@ -944,8 +955,11 @@ class PythonTool:
else:
# Platforms without a usable namespace still receive the same
# alias contract through a conservative source rewrite.
content = _python_with_visible_final_expression(
_replace_workspace_alias(content, agent_cwd())
content = _python_with_configured_import_paths(
_python_with_visible_final_expression(
_replace_workspace_alias(content, agent_cwd())
),
_subproc_env,
)
proc = await asyncio.create_subprocess_exec(
(sys.executable or "python"), "-I", "-c", content,
+49
View File
@@ -270,6 +270,32 @@ class WebSearchTool:
timeout=30,
)
except asyncio.TimeoutError:
# Comprehensive search also downloads several result pages. A
# slow or hostile publisher must not erase the ranked search
# evidence that was already available. Fall back to the metadata
# path so the agent can choose a source and continue with
# web_fetch/private_browser. Keep this bounded independently: the
# abandoned executor thread may still be winding down.
try:
results = await asyncio.wait_for(
loop.run_in_executor(
None,
lambda: searxng_search_results(query, max_pages),
),
timeout=12,
)
text, sources = _format_search_metadata(query, results)
if sources:
output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text
output += "\n\n<!-- SOURCES:" + json.dumps(sources) + " -->"
return {
"output": output,
"exit_code": 0,
"evidence_status": "available",
"degraded_mode": "metadata_after_content_timeout",
}
except Exception:
pass
return {
"error": f"web_search timed out after 30s: {query[:200]}",
"exit_code": 1,
@@ -2190,8 +2216,13 @@ class PrivateBrowserTool:
"batch",
}
_AUTO_SCREENSHOT_ACTIONS = {
"open",
"snapshot",
"batch",
"click",
"fill",
"press",
"scroll",
}
@staticmethod
@@ -2972,6 +3003,15 @@ class PrivateBrowserTool:
for command in commands:
if isinstance(command, list) and command:
action = str(command[0]).strip().lower()
if action == "wait":
# Compact/OpenAI schemas sometimes preserve an omitted
# selector as null and put the timeout in the next slot:
# ["wait", null, 2500]. agent-browser accepts only arrays
# of strings, so recover the intended timeout instead of
# rejecting the whole browser batch.
wait_args = [value for value in command[1:] if value is not None]
normalized.append(["wait", *[str(value) for value in wait_args]])
continue
if action in {"open", "read"} and len(command) >= 2:
candidate_url = str(command[1] or "").strip()
if (
@@ -2985,6 +3025,15 @@ class PrivateBrowserTool:
*command[2:],
])
continue
if action == "read" and not re.match(
r"^(?:https?|file)://", candidate_url, re.IGNORECASE
):
# The top-level read action treats target/selector as
# DOM text extraction. Keep batch semantics identical;
# agent-browser's bare `read h1` instead interprets h1
# as a URL/path and fails before the model can answer.
normalized.append(["get", "text", candidate_url])
continue
if action == "evaluate":
normalized.append(["eval", *command[1:]])
continue