mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-08 16:02:20 +02:00
Consolidate Odysseus agent harness and tool contracts
This commit is contained in:
@@ -560,7 +560,8 @@ async def do_manage_settings(content: str, owner: Optional[str] = None) -> Dict:
|
||||
"hard max": "agent_input_token_hard_max",
|
||||
"token budget cap": "agent_input_token_hard_max",
|
||||
"input budget cap": "agent_input_token_hard_max",
|
||||
"writing style": "email_writing_style", "email writing style": "email_writing_style",
|
||||
"writing style": "document_writing_style", "document writing style": "document_writing_style",
|
||||
"email writing style": "email_writing_style",
|
||||
"reply writing style": "email_writing_style", "email reply writing style": "email_writing_style",
|
||||
}
|
||||
def _resolve(k):
|
||||
|
||||
@@ -11,6 +11,38 @@ from src.upload_handler import reserve_upload_references
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
_DOCUMENT_SEARCH_STOPWORDS = frozenset({
|
||||
'a', 'an', 'and', 'any', 'about', 'document', 'documents', 'for', 'in',
|
||||
'my', 'of', 'on', 'or', 'plans', 'the', 'to',
|
||||
})
|
||||
|
||||
|
||||
def _document_search_tokens(value: str) -> list[str]:
|
||||
return [
|
||||
token for token in re.findall(r'[a-z0-9]+', str(value or '').lower())
|
||||
if token not in _DOCUMENT_SEARCH_STOPWORDS
|
||||
]
|
||||
|
||||
|
||||
def _rank_document_search(docs, search_text: str):
|
||||
"""Prefer phrase/all-term matches, then broaden to any meaningful term."""
|
||||
query = str(search_text or '').strip().lower()
|
||||
terms = _document_search_tokens(query)
|
||||
scored = []
|
||||
for position, doc in enumerate(docs):
|
||||
haystack = ' '.join((
|
||||
str(getattr(doc, 'title', '') or ''),
|
||||
str(getattr(doc, 'current_content', '') or ''),
|
||||
)).lower()
|
||||
haystack_terms = set(_document_search_tokens(haystack))
|
||||
matched = sum(term in haystack_terms for term in terms)
|
||||
strict = bool(query and query in haystack) or bool(terms and matched == len(terms))
|
||||
scored.append((doc, strict, matched, position))
|
||||
strict_matches = [row for row in scored if row[1]]
|
||||
candidates = strict_matches or [row for row in scored if row[2] > 0]
|
||||
return [row[0] for row in sorted(candidates, key=lambda row: (-row[2], row[3]))]
|
||||
|
||||
|
||||
def _missing_document_upload(owner: Optional[str], content: Any) -> Optional[str]:
|
||||
"""Reserve explicit upload URLs before an agent persists document text."""
|
||||
return reserve_upload_references(get_upload_handler(), owner, content)
|
||||
@@ -629,6 +661,12 @@ class UpdateDocumentTool:
|
||||
if is_email_doc:
|
||||
doc.language = "email"
|
||||
|
||||
if new_content == (doc.current_content or ""):
|
||||
return {
|
||||
"error": "No update applied — replacement content is unchanged",
|
||||
"exit_code": 1,
|
||||
}
|
||||
|
||||
missing_id = _missing_document_upload(owner, new_content)
|
||||
if missing_id:
|
||||
return {
|
||||
@@ -761,6 +799,10 @@ class EditDocumentTool:
|
||||
skipped = 0
|
||||
for edit in edits:
|
||||
_find = edit["find"]
|
||||
if _find == edit["replace"]:
|
||||
logger.warning("edit_document: skipping no-op FIND/REPLACE block")
|
||||
skipped += 1
|
||||
continue
|
||||
if _find in updated_content:
|
||||
updated_content = updated_content.replace(_find, edit["replace"], 1)
|
||||
applied += 1
|
||||
@@ -941,10 +983,22 @@ class ManageDocumentTool:
|
||||
search_text = re.sub(
|
||||
r"\s+(?:instead|please)\s*$", "", search_text, flags=re.IGNORECASE
|
||||
).strip()
|
||||
q = q.filter(Document.title.ilike(f"%{search_text}%"))
|
||||
if args.get("language"):
|
||||
q = q.filter(Document.language == args["language"])
|
||||
docs = q.order_by(Document.updated_at.desc()).limit(args.get("limit", 50)).all()
|
||||
requested_limit = args.get("limit", 50)
|
||||
try:
|
||||
requested_limit = max(1, min(int(requested_limit), 200))
|
||||
except (TypeError, ValueError):
|
||||
requested_limit = 50
|
||||
q = q.order_by(Document.updated_at.desc())
|
||||
# A plain listing must not load the entire document library
|
||||
# (including every document body) before applying its limit.
|
||||
if not search_text:
|
||||
q = q.limit(requested_limit)
|
||||
docs = q.all()
|
||||
if search_text:
|
||||
docs = _rank_document_search(docs, search_text)
|
||||
docs = docs[:requested_limit]
|
||||
if not docs:
|
||||
msg = "No documents found" + (f" matching '{search_text}'" if search_text else "") + "."
|
||||
return {"response": msg, "documents": [], "exit_code": 0}
|
||||
|
||||
@@ -20,6 +20,10 @@ _CODENAV_MAX_LINE = 400
|
||||
_STRUCTURED_DOCUMENT_SUFFIXES = frozenset({
|
||||
".doc", ".docx", ".epub", ".pdf", ".pptx", ".xls", ".xlsx",
|
||||
})
|
||||
_BINARY_ARTIFACT_SUFFIXES = _STRUCTURED_DOCUMENT_SUFFIXES | frozenset({
|
||||
".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".mp3", ".mp4", ".ogg",
|
||||
".png", ".wav", ".webm", ".webp", ".zip",
|
||||
})
|
||||
|
||||
|
||||
def _glob_to_regex(pat: str) -> "re.Pattern":
|
||||
@@ -275,6 +279,20 @@ class WriteFileTool:
|
||||
),
|
||||
"exit_code": 1,
|
||||
}
|
||||
# write_file is a UTF-8 text writer. Refuse to silently destroy an
|
||||
# existing PDF, image, archive, or media artifact produced by a
|
||||
# format-aware tool, especially after the agent has verified it.
|
||||
suffix = os.path.splitext(path)[1].casefold()
|
||||
if suffix in _BINARY_ARTIFACT_SUFFIXES:
|
||||
target_existed = os.path.isfile(path)
|
||||
return {
|
||||
"error": (
|
||||
f"write_file: refusing UTF-8 text for binary artifact path {path}. "
|
||||
"Use Python or a format-specific creation tool, then inspect the result."
|
||||
),
|
||||
"exit_code": 1,
|
||||
"binary_artifact_preserved": target_existed,
|
||||
}
|
||||
try:
|
||||
def _write():
|
||||
old = ""
|
||||
|
||||
@@ -433,6 +433,12 @@ class ExtractTextTool:
|
||||
return {"error": "extract_text unknown argument(s): " + ", ".join(unknown), "exit_code": 1}
|
||||
try:
|
||||
raw_path = str(args.get("path") or '')
|
||||
# Some native-schema models serialize a workspace path using the
|
||||
# same URI shape as uploads. This alias grants no extra access:
|
||||
# convert it back to /workspace and let the normal confinement
|
||||
# resolver enforce the active root.
|
||||
if raw_path.startswith('odysseus://workspace/'):
|
||||
raw_path = '/workspace/' + raw_path[len('odysseus://workspace/'):]
|
||||
if raw_path.startswith('odysseus://'):
|
||||
# Upload access is independent of a filesystem workspace and
|
||||
# must never inherit an administrator's cross-owner override.
|
||||
@@ -450,8 +456,9 @@ class ExtractTextTool:
|
||||
path = _resolve_media_path(raw_path, tool_name="extract_text")
|
||||
except ValueError as exc:
|
||||
return {"error": str(exc), "exit_code": 1}
|
||||
if path.suffix.casefold() not in _IMAGE_SUFFIXES:
|
||||
return {"error": "extract_text currently supports local image files", "exit_code": 1}
|
||||
suffix = path.suffix.casefold()
|
||||
if suffix not in _IMAGE_SUFFIXES | _PDF_SUFFIXES:
|
||||
return {"error": "extract_text supports local image and PDF files", "exit_code": 1}
|
||||
mode = str(args.get("mode") or "all").strip().casefold()
|
||||
try:
|
||||
minimum, maximum = float(args.get("min_confidence", .5)), int(args.get("max_results", 512))
|
||||
@@ -461,7 +468,56 @@ class ExtractTextTool:
|
||||
return {"error": "invalid extract_text mode or bounds", "exit_code": 1}
|
||||
try:
|
||||
from .ocr_engine import extract_image_text
|
||||
evidence = await asyncio.to_thread(extract_image_text, path, include_layout=bool(args.get("include_layout", False)), numeric_only=mode == "numbers", min_confidence=minimum, max_results=maximum)
|
||||
if suffix in _PDF_SUFFIXES:
|
||||
def _extract_pdf_pages():
|
||||
try:
|
||||
import pypdfium2 as pdfium
|
||||
except ImportError as exc:
|
||||
raise RuntimeError(
|
||||
"PDF OCR requires the optional pypdfium2 package"
|
||||
) from exc
|
||||
document = pdfium.PdfDocument(str(path))
|
||||
page_count = len(document)
|
||||
lines, accepted = [], 0
|
||||
# Keep one OCR call bounded while covering ordinary
|
||||
# documents completely. Larger PDFs can be inspected in
|
||||
# page ranges with inspect_media.
|
||||
rendered_count = min(page_count, 12)
|
||||
with tempfile.TemporaryDirectory(prefix="odysseus-pdf-ocr-") as temp_dir:
|
||||
for index in range(rendered_count):
|
||||
rendered = document[index].render(scale=2.0).to_pil().convert("RGB")
|
||||
image_path = Path(temp_dir) / f"page-{index + 1}.png"
|
||||
rendered.save(image_path, "PNG")
|
||||
remaining = max(1, maximum - len(lines))
|
||||
page_evidence = extract_image_text(
|
||||
image_path,
|
||||
include_layout=bool(args.get("include_layout", False)),
|
||||
numeric_only=mode == "numbers",
|
||||
min_confidence=minimum,
|
||||
max_results=remaining,
|
||||
)
|
||||
accepted += int(page_evidence.get("count") or 0)
|
||||
for line in page_evidence.get("lines") or []:
|
||||
if len(lines) >= maximum:
|
||||
break
|
||||
lines.append({"page": index + 1, **line})
|
||||
return {
|
||||
"legend": {
|
||||
"page": "one-based PDF page",
|
||||
"t": "text",
|
||||
"p": "confidence",
|
||||
"xy": "pixel center",
|
||||
},
|
||||
"page_count": page_count,
|
||||
"pages_processed": rendered_count,
|
||||
"count": accepted,
|
||||
"returned": len(lines),
|
||||
"truncated": accepted > len(lines) or page_count > rendered_count,
|
||||
"lines": lines,
|
||||
}
|
||||
evidence = await asyncio.to_thread(_extract_pdf_pages)
|
||||
else:
|
||||
evidence = await asyncio.to_thread(extract_image_text, path, include_layout=bool(args.get("include_layout", False)), numeric_only=mode == "numbers", min_confidence=minimum, max_results=maximum)
|
||||
except Exception as exc:
|
||||
return {"error": f"extract_text failed: {exc}", "exit_code": 1}
|
||||
return {"output": json.dumps(evidence, ensure_ascii=False, separators=(",", ":")), "exit_code": 0, "ocr": evidence}
|
||||
@@ -715,9 +771,16 @@ class InspectMediaTool:
|
||||
suffix = path.suffix.lower()
|
||||
if suffix in _SVG_SUFFIXES:
|
||||
renderer = shutil.which("rsvg-convert")
|
||||
renderer_kind = "rsvg"
|
||||
if not renderer:
|
||||
renderer = shutil.which("convert")
|
||||
renderer_kind = "imagemagick"
|
||||
if not renderer:
|
||||
return {
|
||||
"error": "inspect_media SVG rendering requires rsvg-convert",
|
||||
"error": (
|
||||
"inspect_media SVG rendering requires rsvg-convert "
|
||||
"or ImageMagick convert"
|
||||
),
|
||||
"exit_code": 1,
|
||||
}
|
||||
raw_output = str(args.get("output_path") or "").strip()
|
||||
@@ -740,7 +803,11 @@ class InspectMediaTool:
|
||||
output = Path(temporary.name)
|
||||
rendered = await asyncio.to_thread(
|
||||
_run,
|
||||
[renderer, "--output", str(output), str(path)],
|
||||
(
|
||||
[renderer, "--output", str(output), str(path)]
|
||||
if renderer_kind == "rsvg"
|
||||
else [renderer, str(path), str(output)]
|
||||
),
|
||||
60,
|
||||
)
|
||||
if rendered.returncode != 0 or not output.is_file() or output.stat().st_size == 0:
|
||||
|
||||
@@ -133,6 +133,62 @@ async def list_models(content: str, session_id: Optional[str] = None, owner: Opt
|
||||
|
||||
keyword = content.strip().lower() if content.strip() else None
|
||||
|
||||
# ``list_models`` historically treated every filter as a literal model-ID
|
||||
# substring. For recommendation terms that produced an empty catalog even
|
||||
# though Odysseus already has a hardware detector and fit ranker. Preserve
|
||||
# the catalog behavior for real model/provider filters, but give these
|
||||
# semantic filters their expected read-only meaning.
|
||||
if keyword in {
|
||||
"recommended", "recommendation", "recommendations",
|
||||
"compatible", "hardware", "hardware fit", "best fit",
|
||||
}:
|
||||
from src.tools.system import do_app_api
|
||||
fit_result = await do_app_api(json.dumps({
|
||||
"action": "call",
|
||||
"method": "GET",
|
||||
"path": "/api/hwfit/models",
|
||||
"query": {"fit_only": "true", "limit": 5, "sort": "fit"},
|
||||
}), owner=owner)
|
||||
payload = fit_result.get("json") if isinstance(fit_result, dict) else None
|
||||
system = payload.get("system") if isinstance(payload, dict) else None
|
||||
models = payload.get("models") if isinstance(payload, dict) else None
|
||||
if isinstance(system, dict) and isinstance(models, list):
|
||||
gpu = system.get("gpu_name") or "No GPU detected"
|
||||
vram = system.get("gpu_vram_gb")
|
||||
count = system.get("gpu_count")
|
||||
backend = system.get("backend") or "unknown"
|
||||
lines = [
|
||||
"Detected hardware:",
|
||||
f"- GPU: {gpu}; count={count}; total VRAM={vram} GB; backend={backend}",
|
||||
f"- CPU: {system.get('cpu_name') or 'unknown'}; RAM={system.get('total_ram_gb')} GB",
|
||||
"Ranked compatible models:",
|
||||
]
|
||||
compact_models = []
|
||||
for model_row in models[:5]:
|
||||
if not isinstance(model_row, dict):
|
||||
continue
|
||||
compact = {
|
||||
key: model_row.get(key)
|
||||
for key in (
|
||||
"name", "parameter_count", "quant", "required_gb",
|
||||
"fit_level", "run_mode", "speed_tps", "score", "context",
|
||||
)
|
||||
}
|
||||
compact_models.append(compact)
|
||||
lines.append(
|
||||
"- {name}: params={parameter_count}, quant={quant}, required={required_gb} GB, "
|
||||
"fit={fit_level}, mode={run_mode}, speed={speed_tps} tok/s, score={score}, context={context}".format(
|
||||
**compact
|
||||
)
|
||||
)
|
||||
return {
|
||||
"output": "\n".join(lines),
|
||||
"system": system,
|
||||
"models": compact_models,
|
||||
"exit_code": 0,
|
||||
}
|
||||
return fit_result
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
query = db.query(ModelEndpoint).filter(ModelEndpoint.is_enabled == True)
|
||||
|
||||
@@ -874,6 +874,15 @@ def _python_with_visible_final_expression(content: str) -> str:
|
||||
return ast.unparse(tree)
|
||||
|
||||
|
||||
def _python_with_configured_import_paths(content: str, env: dict | None) -> str:
|
||||
"""Expose only explicitly configured package roots under Python ``-I``."""
|
||||
raw = str((env or {}).get("ODYSSEUS_PYTHON_TOOL_SITE_PACKAGES", ""))
|
||||
paths = [item for item in raw.split(os.pathsep) if item and os.path.isabs(item)]
|
||||
if not paths:
|
||||
return content
|
||||
return f"import site\n[site.addsitedir(path) for path in {paths!r}]\nexec(compile({content!r}, '<odysseus-python-tool>', 'exec'))"
|
||||
|
||||
|
||||
class PythonTool:
|
||||
async def execute(self, content: str, ctx: dict) -> dict:
|
||||
from src.tool_execution import agent_cwd, _truncate
|
||||
@@ -917,7 +926,9 @@ class PythonTool:
|
||||
# process-global `/workspace` symlink would break concurrent tasks.
|
||||
# Give Python the same per-task namespace Bash receives so both inline
|
||||
# code and loaded scripts see the stable virtual workspace root.
|
||||
namespaced_content = _python_with_visible_final_expression(content)
|
||||
namespaced_content = _python_with_configured_import_paths(
|
||||
_python_with_visible_final_expression(content), _subproc_env
|
||||
)
|
||||
python_command = shlex.join((sys.executable or "python", "-I", "-c", namespaced_content))
|
||||
# Code that explicitly uses the public /workspace path runs inside a
|
||||
# namespace whose stable cwd is that same bind. Host workspaces under
|
||||
@@ -944,8 +955,11 @@ class PythonTool:
|
||||
else:
|
||||
# Platforms without a usable namespace still receive the same
|
||||
# alias contract through a conservative source rewrite.
|
||||
content = _python_with_visible_final_expression(
|
||||
_replace_workspace_alias(content, agent_cwd())
|
||||
content = _python_with_configured_import_paths(
|
||||
_python_with_visible_final_expression(
|
||||
_replace_workspace_alias(content, agent_cwd())
|
||||
),
|
||||
_subproc_env,
|
||||
)
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
(sys.executable or "python"), "-I", "-c", content,
|
||||
|
||||
@@ -270,6 +270,32 @@ class WebSearchTool:
|
||||
timeout=30,
|
||||
)
|
||||
except asyncio.TimeoutError:
|
||||
# Comprehensive search also downloads several result pages. A
|
||||
# slow or hostile publisher must not erase the ranked search
|
||||
# evidence that was already available. Fall back to the metadata
|
||||
# path so the agent can choose a source and continue with
|
||||
# web_fetch/private_browser. Keep this bounded independently: the
|
||||
# abandoned executor thread may still be winding down.
|
||||
try:
|
||||
results = await asyncio.wait_for(
|
||||
loop.run_in_executor(
|
||||
None,
|
||||
lambda: searxng_search_results(query, max_pages),
|
||||
),
|
||||
timeout=12,
|
||||
)
|
||||
text, sources = _format_search_metadata(query, results)
|
||||
if sources:
|
||||
output = text[:MAX_OUTPUT_CHARS] if len(text) > MAX_OUTPUT_CHARS else text
|
||||
output += "\n\n<!-- SOURCES:" + json.dumps(sources) + " -->"
|
||||
return {
|
||||
"output": output,
|
||||
"exit_code": 0,
|
||||
"evidence_status": "available",
|
||||
"degraded_mode": "metadata_after_content_timeout",
|
||||
}
|
||||
except Exception:
|
||||
pass
|
||||
return {
|
||||
"error": f"web_search timed out after 30s: {query[:200]}",
|
||||
"exit_code": 1,
|
||||
@@ -2190,8 +2216,13 @@ class PrivateBrowserTool:
|
||||
"batch",
|
||||
}
|
||||
_AUTO_SCREENSHOT_ACTIONS = {
|
||||
"open",
|
||||
"snapshot",
|
||||
"batch",
|
||||
"click",
|
||||
"fill",
|
||||
"press",
|
||||
"scroll",
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
@@ -2972,6 +3003,15 @@ class PrivateBrowserTool:
|
||||
for command in commands:
|
||||
if isinstance(command, list) and command:
|
||||
action = str(command[0]).strip().lower()
|
||||
if action == "wait":
|
||||
# Compact/OpenAI schemas sometimes preserve an omitted
|
||||
# selector as null and put the timeout in the next slot:
|
||||
# ["wait", null, 2500]. agent-browser accepts only arrays
|
||||
# of strings, so recover the intended timeout instead of
|
||||
# rejecting the whole browser batch.
|
||||
wait_args = [value for value in command[1:] if value is not None]
|
||||
normalized.append(["wait", *[str(value) for value in wait_args]])
|
||||
continue
|
||||
if action in {"open", "read"} and len(command) >= 2:
|
||||
candidate_url = str(command[1] or "").strip()
|
||||
if (
|
||||
@@ -2985,6 +3025,15 @@ class PrivateBrowserTool:
|
||||
*command[2:],
|
||||
])
|
||||
continue
|
||||
if action == "read" and not re.match(
|
||||
r"^(?:https?|file)://", candidate_url, re.IGNORECASE
|
||||
):
|
||||
# The top-level read action treats target/selector as
|
||||
# DOM text extraction. Keep batch semantics identical;
|
||||
# agent-browser's bare `read h1` instead interprets h1
|
||||
# as a URL/path and fails before the model can answer.
|
||||
normalized.append(["get", "text", candidate_url])
|
||||
continue
|
||||
if action == "evaluate":
|
||||
normalized.append(["eval", *command[1:]])
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user