mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-07 15:32:21 +02:00
Squash Odysseus development history
This commit is contained in:
+236
-4
@@ -22,6 +22,18 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
RESEARCH_DATA_DIR = Path(DEEP_RESEARCH_DIR)
|
||||
_RESEARCH_SESSION_ID_RE = re.compile(r"^[A-Za-z0-9-]{1,128}$")
|
||||
_SEARCH_CONTINUATIONS = {
|
||||
"search",
|
||||
"search this",
|
||||
"can you search",
|
||||
"can you search this",
|
||||
"please search",
|
||||
"look it up",
|
||||
"look this up",
|
||||
"web search",
|
||||
"use web",
|
||||
"search online",
|
||||
}
|
||||
|
||||
|
||||
def _bounded_int(value, *, default: int, minimum: int, maximum: int) -> int:
|
||||
@@ -122,14 +134,18 @@ class ResearchHandler:
|
||||
def _normalize(text: str) -> str:
|
||||
return (text or "").strip().lower().strip("!.? ")
|
||||
|
||||
def _is_continuation(text: str) -> bool:
|
||||
normalized = re.sub(r"\s+", " ", _normalize(text))
|
||||
return normalized in _AFFIRMATIONS or normalized in _SEARCH_CONTINUATIONS
|
||||
|
||||
def _fallback() -> str:
|
||||
normalized = _normalize(latest_message)
|
||||
if normalized and normalized not in _AFFIRMATIONS:
|
||||
if normalized and not _is_continuation(latest_message):
|
||||
return latest_message # short or long, it's a real topic
|
||||
# Affirmation, or empty/punctuation-only: use the original ask.
|
||||
for m in history:
|
||||
c = (m.content or "").strip()
|
||||
if m.role == "user" and c and _normalize(c) not in _AFFIRMATIONS:
|
||||
if m.role == "user" and c and not _is_continuation(c):
|
||||
return c
|
||||
return latest_message
|
||||
|
||||
@@ -302,6 +318,7 @@ class ResearchHandler:
|
||||
"result": None,
|
||||
"started_at": time.time(),
|
||||
"category": category,
|
||||
"mode": "research",
|
||||
# SECURITY: track ownership so all reads / saves can filter by user.
|
||||
"owner": owner or "",
|
||||
}
|
||||
@@ -337,6 +354,7 @@ class ResearchHandler:
|
||||
max_rounds=max_rounds,
|
||||
search_provider=search_provider,
|
||||
category=category,
|
||||
session_id=session_id,
|
||||
extraction_timeout=extraction_timeout,
|
||||
extraction_concurrency=extraction_concurrency,
|
||||
),
|
||||
@@ -413,6 +431,12 @@ class ResearchHandler:
|
||||
"progress": entry["progress"],
|
||||
"query": entry["query"],
|
||||
"started_at": entry["started_at"],
|
||||
"category": (
|
||||
getattr(entry.get("researcher"), "category", None)
|
||||
or entry.get("category")
|
||||
or ""
|
||||
),
|
||||
"mode": entry.get("mode") or "research",
|
||||
}
|
||||
# avg_duration is a historical figure over completed reports on
|
||||
# disk; get_avg_duration() globs and JSON-parses the whole research
|
||||
@@ -439,11 +463,33 @@ class ResearchHandler:
|
||||
"progress": {},
|
||||
"query": data.get("query", ""),
|
||||
"started_at": data.get("started_at", 0),
|
||||
"category": data.get("category") or "",
|
||||
"mode": data.get("mode") or "research",
|
||||
}
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
def get_category(self, session_id: str) -> str:
|
||||
"""Return the requested or auto-resolved report format."""
|
||||
if session_id in self._active_tasks:
|
||||
entry = self._active_tasks[session_id]
|
||||
researcher = entry.get("researcher")
|
||||
return str(
|
||||
getattr(researcher, "category", None)
|
||||
or entry.get("category")
|
||||
or ""
|
||||
)
|
||||
data = self._get_session_json(session_id)
|
||||
return str(data.get("category") or "") if isinstance(data, dict) else ""
|
||||
|
||||
def get_mode(self, session_id: str) -> str:
|
||||
"""Return whether the task performs research or a model-only explanation."""
|
||||
if session_id in self._active_tasks:
|
||||
return str(self._active_tasks[session_id].get("mode") or "research")
|
||||
data = self._get_session_json(session_id)
|
||||
return str(data.get("mode") or "research") if isinstance(data, dict) else "research"
|
||||
|
||||
def cancel_research(self, session_id: str) -> bool:
|
||||
"""Cancel running research for a session."""
|
||||
if session_id not in self._active_tasks:
|
||||
@@ -521,6 +567,87 @@ class ResearchHandler:
|
||||
logger.warning(f"Failed to read raw findings for {session_id}: {e}")
|
||||
return None
|
||||
|
||||
def get_analyzed_urls(self, session_id: str) -> Optional[list]:
|
||||
"""Get all analyzed URLs, including pages that did not yield findings."""
|
||||
if session_id in self._active_tasks:
|
||||
researcher = self._active_tasks[session_id].get("researcher")
|
||||
if researcher:
|
||||
return list(getattr(researcher, "analyzed_urls", []) or [])
|
||||
data = self._get_session_json(session_id)
|
||||
if isinstance(data, dict):
|
||||
return data.get("analyzed_urls")
|
||||
return None
|
||||
|
||||
def get_source_state(self, session_id: str) -> str:
|
||||
"""Get compact source-quality/gap state for UI/debug display."""
|
||||
if session_id in self._active_tasks:
|
||||
researcher = self._active_tasks[session_id].get("researcher")
|
||||
if researcher:
|
||||
try:
|
||||
return researcher._source_state_summary()
|
||||
except Exception:
|
||||
return ""
|
||||
data = self._get_session_json(session_id)
|
||||
if isinstance(data, dict):
|
||||
return str(data.get("source_state") or "")
|
||||
return ""
|
||||
|
||||
def get_source_coverage(self, session_id: str) -> dict:
|
||||
"""Get machine-readable source coverage stats for UI/debug display."""
|
||||
if session_id in self._active_tasks:
|
||||
researcher = self._active_tasks[session_id].get("researcher")
|
||||
if researcher:
|
||||
try:
|
||||
coverage = researcher._source_coverage()
|
||||
return coverage if isinstance(coverage, dict) else {}
|
||||
except Exception:
|
||||
return {}
|
||||
data = self._get_session_json(session_id)
|
||||
if isinstance(data, dict):
|
||||
coverage = data.get("source_coverage")
|
||||
return coverage if isinstance(coverage, dict) else {}
|
||||
return {}
|
||||
|
||||
def get_navigation_trace(self, session_id: str) -> list:
|
||||
"""Get bounded research navigation/tool observations for debugging."""
|
||||
if session_id in self._active_tasks:
|
||||
researcher = self._active_tasks[session_id].get("researcher")
|
||||
if researcher:
|
||||
trace = getattr(researcher, "navigation_trace", []) or []
|
||||
return list(trace) if isinstance(trace, list) else []
|
||||
data = self._get_session_json(session_id)
|
||||
if isinstance(data, dict):
|
||||
trace = data.get("navigation_trace")
|
||||
return trace if isinstance(trace, list) else []
|
||||
return []
|
||||
|
||||
def get_action_trace(self, session_id: str) -> list:
|
||||
"""Get bounded research planner actions for debugging."""
|
||||
if session_id in self._active_tasks:
|
||||
researcher = self._active_tasks[session_id].get("researcher")
|
||||
if researcher:
|
||||
trace = getattr(researcher, "action_trace", []) or []
|
||||
return list(trace) if isinstance(trace, list) else []
|
||||
data = self._get_session_json(session_id)
|
||||
if isinstance(data, dict):
|
||||
trace = data.get("action_trace")
|
||||
return trace if isinstance(trace, list) else []
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def _source_metadata(f: dict) -> dict:
|
||||
meta = {}
|
||||
for key in ("retrieval", "source_kind", "source_reason"):
|
||||
value = f.get(key)
|
||||
if value:
|
||||
meta[key] = value
|
||||
try:
|
||||
score = int(f.get("source_score"))
|
||||
meta["source_score"] = score
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
return meta
|
||||
|
||||
@staticmethod
|
||||
def _extract_sources(findings: list) -> list:
|
||||
"""Extract deduplicated [{url, title}] from findings, filtering low-quality ones."""
|
||||
@@ -538,6 +665,7 @@ class ResearchHandler:
|
||||
og_img = f.get("og_image", "")
|
||||
if og_img:
|
||||
entry["image"] = og_img
|
||||
entry.update(ResearchHandler._source_metadata(f))
|
||||
sources.append(entry)
|
||||
return sources
|
||||
|
||||
@@ -555,7 +683,9 @@ class ResearchHandler:
|
||||
evidence = f.get("evidence", "")
|
||||
content = summary if summary else (evidence[:2000] if evidence else "")
|
||||
if url and content and not is_low_quality(content):
|
||||
items.append({"url": url, "title": title, "summary": content})
|
||||
item = {"url": url, "title": title, "summary": content}
|
||||
item.update(ResearchHandler._source_metadata(f))
|
||||
items.append(item)
|
||||
return items
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to extract raw findings: {e}")
|
||||
@@ -613,6 +743,24 @@ class ResearchHandler:
|
||||
sources = self._extract_sources(researcher.findings)
|
||||
raw_findings = self._extract_raw_findings(researcher.findings)
|
||||
entry["sources"] = sources
|
||||
source_state = ""
|
||||
source_coverage = {}
|
||||
analyzed_urls = []
|
||||
navigation_trace = []
|
||||
action_trace = []
|
||||
if researcher:
|
||||
analyzed_urls = list(getattr(researcher, "analyzed_urls", []) or [])
|
||||
trace = getattr(researcher, "navigation_trace", []) or []
|
||||
navigation_trace = list(trace) if isinstance(trace, list) else []
|
||||
planned = getattr(researcher, "action_trace", []) or []
|
||||
action_trace = list(planned) if isinstance(planned, list) else []
|
||||
try:
|
||||
source_state = researcher._source_state_summary()
|
||||
coverage = researcher._source_coverage()
|
||||
source_coverage = coverage if isinstance(coverage, dict) else {}
|
||||
except Exception:
|
||||
source_state = ""
|
||||
source_coverage = {}
|
||||
|
||||
data = {
|
||||
"query": entry["query"],
|
||||
@@ -621,8 +769,14 @@ class ResearchHandler:
|
||||
"raw_report": entry.get("raw_report", ""),
|
||||
"sources": sources,
|
||||
"raw_findings": raw_findings,
|
||||
"analyzed_urls": analyzed_urls,
|
||||
"source_state": source_state,
|
||||
"source_coverage": source_coverage,
|
||||
"navigation_trace": navigation_trace,
|
||||
"action_trace": action_trace,
|
||||
"stats": entry.get("stats"),
|
||||
"category": entry.get("category"),
|
||||
"mode": entry.get("mode") or "research",
|
||||
"started_at": entry["started_at"],
|
||||
"completed_at": time.time(),
|
||||
# SECURITY: stamp owner so route handlers can filter by user.
|
||||
@@ -650,6 +804,70 @@ class ResearchHandler:
|
||||
pass
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _format_trace_value(value) -> str:
|
||||
text = re.sub(r"\s+", " ", str(value or "")).strip()
|
||||
return text.replace("`", "'")
|
||||
|
||||
@staticmethod
|
||||
def _research_diagnostics_markdown(data: dict) -> str:
|
||||
"""Collapsed debug section for visual research reports."""
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
action_trace = data.get("action_trace") if isinstance(data.get("action_trace"), list) else []
|
||||
navigation_trace = data.get("navigation_trace") if isinstance(data.get("navigation_trace"), list) else []
|
||||
source_state = str(data.get("source_state") or "").strip()
|
||||
if not action_trace and not navigation_trace and not source_state:
|
||||
return ""
|
||||
|
||||
lines = [
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
'<details markdown="1">',
|
||||
"<summary>Research trace</summary>",
|
||||
"",
|
||||
]
|
||||
if action_trace:
|
||||
lines.extend(["### Planned Actions", ""])
|
||||
for item in action_trace[-20:]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
round_label = f"Round {item.get('round')}" if item.get("round") else "Round"
|
||||
source = ResearchHandler._format_trace_value(item.get("source") or "planner")
|
||||
tool = ResearchHandler._format_trace_value(item.get("tool") or "tool")
|
||||
requested_by = ResearchHandler._format_trace_value(item.get("requested_by") or "")
|
||||
target = ResearchHandler._format_trace_value(item.get("query") or item.get("url") or "")
|
||||
if item.get("status") == "skipped":
|
||||
reason = ResearchHandler._format_trace_value(item.get("reason") or "skipped")
|
||||
lines.append(f"- **{round_label}** `{source}` skipped `{tool}` {target} — {reason}")
|
||||
else:
|
||||
alias = f" via `{requested_by}`" if requested_by and requested_by != tool else ""
|
||||
lines.append(f"- **{round_label}** `{source}` -> `{tool}`{alias} {target}")
|
||||
lines.append("")
|
||||
if navigation_trace:
|
||||
lines.extend(["### Navigation", ""])
|
||||
for item in navigation_trace[-20:]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
tool = ResearchHandler._format_trace_value(item.get("tool") or "tool")
|
||||
status = ResearchHandler._format_trace_value(item.get("status") or "unknown")
|
||||
target = ResearchHandler._format_trace_value(item.get("query") or item.get("title") or item.get("url") or "")
|
||||
meta = []
|
||||
if item.get("results") is not None:
|
||||
meta.append(f"{item.get('results')} results")
|
||||
if item.get("source_kind"):
|
||||
meta.append(ResearchHandler._format_trace_value(item.get("source_kind")))
|
||||
if item.get("source_score") is not None:
|
||||
meta.append(f"{item.get('source_score')}/100")
|
||||
suffix = f" ({'; '.join(meta)})" if meta else ""
|
||||
lines.append(f"- `{tool}` {target} -> **{status}**{suffix}")
|
||||
lines.append("")
|
||||
if source_state:
|
||||
lines.extend(["### Source State", "", "```text", source_state[:2000], "```", ""])
|
||||
lines.append("</details>")
|
||||
return "\n".join(lines)
|
||||
|
||||
def get_report_html(self, session_id: str) -> Optional[str]:
|
||||
"""Generate the visual HTML report for a session (always fresh from JSON)."""
|
||||
json_path = _research_json_path(session_id)
|
||||
@@ -664,6 +882,9 @@ class ResearchHandler:
|
||||
|
||||
data = json.loads(json_path.read_text(encoding="utf-8"))
|
||||
report_md = data.get("raw_report") or data.get("result", "")
|
||||
diagnostics = self._research_diagnostics_markdown(data)
|
||||
if diagnostics:
|
||||
report_md = f"{report_md.rstrip()}\n{diagnostics}"
|
||||
html_content = generate_visual_report(
|
||||
question=data.get("query", ""),
|
||||
report_markdown=report_md,
|
||||
@@ -754,6 +975,7 @@ class ResearchHandler:
|
||||
category: str = None,
|
||||
extraction_timeout: int = None,
|
||||
extraction_concurrency: int = None,
|
||||
session_id: str = "",
|
||||
) -> str:
|
||||
"""
|
||||
Run iterative deep research using the LLM-in-the-loop DeepResearcher.
|
||||
@@ -771,6 +993,12 @@ class ResearchHandler:
|
||||
Returns:
|
||||
Formatted research report with expandable section and summary
|
||||
"""
|
||||
if max_rounds < 0:
|
||||
raise ValueError("max_rounds must be 0 or greater")
|
||||
allowed_categories = {None, "product", "comparison", "howto", "factcheck"}
|
||||
if category not in allowed_categories:
|
||||
raise ValueError(f"Unsupported research category: {category}")
|
||||
|
||||
is_continuation = bool(prior_report)
|
||||
logger.info(f"{'Continuing' if is_continuation else 'Starting'} IterResearch Deep Research")
|
||||
logger.info(f"Query: {query}")
|
||||
@@ -779,7 +1007,7 @@ class ResearchHandler:
|
||||
if is_continuation:
|
||||
logger.info(f"Prior: {len(prior_findings or [])} findings, {len(prior_urls or set())} URLs")
|
||||
|
||||
# Probe the endpoint before committing to a long research run
|
||||
# Probe the endpoint before committing to a long research run.
|
||||
if progress_callback:
|
||||
progress_callback({"phase": "probing", "model": llm_model})
|
||||
await self._probe_endpoint(llm_endpoint, llm_model, llm_headers)
|
||||
@@ -829,6 +1057,7 @@ class ResearchHandler:
|
||||
progress_callback=progress_callback,
|
||||
search_provider=search_provider,
|
||||
category=category,
|
||||
session_id=session_id,
|
||||
)
|
||||
if _task_entry is not None:
|
||||
_task_entry["researcher"] = researcher
|
||||
@@ -851,6 +1080,9 @@ class ResearchHandler:
|
||||
if _task_entry is not None:
|
||||
_task_entry["raw_report"] = strip_thinking(report)
|
||||
_task_entry["stats"] = stats
|
||||
# Auto classification happens inside DeepResearcher. Keep the
|
||||
# resolved format on the task so it survives every UI path.
|
||||
_task_entry["category"] = researcher.category or category
|
||||
|
||||
return self._format_research_report(query, report, stats, elapsed)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user