mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-07 07:22:21 +02:00
Squash Odysseus development history
This commit is contained in:
+207
-32
@@ -10,9 +10,10 @@ import logging
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from src.model_context import get_context_length, estimate_tokens
|
||||
from src.model_context import estimate_text_tokens, get_context_length, estimate_tokens
|
||||
from src.llm_core import llm_call_async
|
||||
from src.endpoint_resolver import resolve_endpoint
|
||||
from src.settings import get_setting
|
||||
from core.models import ChatMessage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -37,10 +38,72 @@ def _content_as_text(content: Any) -> str:
|
||||
return ""
|
||||
|
||||
|
||||
_MULTIMODAL_IMAGE_TYPES = {"image_url", "input_image", "image"}
|
||||
|
||||
|
||||
def prune_multimodal_images(
|
||||
messages: List[Dict],
|
||||
*,
|
||||
max_images: int = 8,
|
||||
) -> List[Dict]:
|
||||
"""Keep uniformly sampled visual blocks across a multimodal history."""
|
||||
|
||||
image_locations = []
|
||||
for message_index, message in enumerate(messages):
|
||||
content = message.get("content") if isinstance(message, dict) else None
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
for content_index, item in enumerate(content):
|
||||
if isinstance(item, dict) and item.get("type") in _MULTIMODAL_IMAGE_TYPES:
|
||||
image_locations.append((message_index, content_index))
|
||||
|
||||
limit = max(0, int(max_images))
|
||||
if len(image_locations) <= limit:
|
||||
return list(messages)
|
||||
if limit == 0:
|
||||
selected = set()
|
||||
elif limit == 1:
|
||||
selected = {image_locations[-1]}
|
||||
else:
|
||||
last = len(image_locations) - 1
|
||||
selected = {
|
||||
image_locations[round(index * last / (limit - 1))]
|
||||
for index in range(limit)
|
||||
}
|
||||
|
||||
pruned = []
|
||||
for message_index, message in enumerate(messages):
|
||||
if not isinstance(message, dict):
|
||||
pruned.append(message)
|
||||
continue
|
||||
cloned = dict(message)
|
||||
content = message.get("content")
|
||||
if isinstance(content, list):
|
||||
cloned["content"] = [
|
||||
item for content_index, item in enumerate(content)
|
||||
if not (
|
||||
isinstance(item, dict)
|
||||
and item.get("type") in _MULTIMODAL_IMAGE_TYPES
|
||||
and (message_index, content_index) not in selected
|
||||
)
|
||||
]
|
||||
pruned.append(cloned)
|
||||
return pruned
|
||||
|
||||
|
||||
COMPACT_THRESHOLD = 0.85 # Trigger compaction at 85% of context window
|
||||
SUMMARY_MAX_TOKENS = 1024
|
||||
SMALL_CONTEXT_LIMIT = 8192 # Models with context <= this get aggressive trimming
|
||||
|
||||
|
||||
def auto_compact_threshold_percent() -> int:
|
||||
"""Configured auto-compaction threshold, clamped to a sane UI range."""
|
||||
try:
|
||||
value = int(get_setting("auto_compact_threshold_percent", int(COMPACT_THRESHOLD * 100)) or 85)
|
||||
except (TypeError, ValueError):
|
||||
value = int(COMPACT_THRESHOLD * 100)
|
||||
return max(50, min(95, value))
|
||||
|
||||
# Cursor-style self-summarization prompt — produces structured, dense summaries
|
||||
SELF_SUMMARY_SYSTEM_PROMPT = """You are summarizing a conversation to preserve context after compaction. Produce a structured summary that lets the conversation continue seamlessly.
|
||||
|
||||
@@ -128,7 +191,7 @@ def _sanitize_tool_messages(msgs: List[Dict]) -> List[Dict]:
|
||||
def _message_text_token_estimate(text: str) -> int:
|
||||
if not isinstance(text, str):
|
||||
return 4
|
||||
return int(len(text) * 0.3) + 4
|
||||
return estimate_text_tokens(text) + 4
|
||||
|
||||
|
||||
def _truncate_text_to_token_budget(text: str, token_budget: int) -> str:
|
||||
@@ -141,19 +204,36 @@ def _truncate_text_to_token_budget(text: str, token_budget: int) -> str:
|
||||
# string rather than the raw non-string (which would move the crash
|
||||
# into the caller that concatenates/measures the result).
|
||||
return ""
|
||||
# Match src.model_context.estimate_tokens' rough chars * 0.3 estimate.
|
||||
max_chars = max(200, int((token_budget - 16) / 0.3))
|
||||
if len(text) <= max_chars:
|
||||
if estimate_text_tokens(text) <= token_budget - 16:
|
||||
return text
|
||||
|
||||
notice = (
|
||||
"\n\n[Notice: the pasted message was too large for this model's context "
|
||||
"window, so Odysseus kept the beginning and end.]"
|
||||
)
|
||||
keep_chars = max(200, max_chars - len(notice))
|
||||
head_len = max(100, int(keep_chars * 0.7))
|
||||
tail_len = max(80, keep_chars - head_len)
|
||||
return text[:head_len].rstrip() + notice + "\n\n" + text[-tail_len:].lstrip()
|
||||
# Binary replacement characters and dense scripts can approach one token
|
||||
# per character, while ASCII prose is closer to the historical 0.3 ratio.
|
||||
# Find the largest head/tail sample that fits the shared estimator instead
|
||||
# of guessing a character count from one language family.
|
||||
target = max(64, token_budget - estimate_text_tokens(notice) - 16)
|
||||
low, high = 1, len(text)
|
||||
best = 1
|
||||
while low <= high:
|
||||
keep_chars = (low + high) // 2
|
||||
head_len = max(1, int(keep_chars * 0.7))
|
||||
tail_len = max(0, keep_chars - head_len)
|
||||
sample = text[:head_len]
|
||||
if tail_len:
|
||||
sample += text[-tail_len:]
|
||||
if estimate_text_tokens(sample) <= target:
|
||||
best = keep_chars
|
||||
low = keep_chars + 1
|
||||
else:
|
||||
high = keep_chars - 1
|
||||
head_len = max(1, int(best * 0.7))
|
||||
tail_len = max(0, best - head_len)
|
||||
tail = text[-tail_len:].lstrip() if tail_len else ""
|
||||
return text[:head_len].rstrip() + notice + ("\n\n" + tail if tail else "")
|
||||
|
||||
|
||||
def _truncate_tool_call_args(msg: Dict[str, Any], token_budget: int) -> Dict[str, Any]:
|
||||
@@ -180,7 +260,7 @@ def _truncate_tool_call_args(msg: Dict[str, Any], token_budget: int) -> Dict[str
|
||||
for tc in tool_calls:
|
||||
fn = tc.get("function") if isinstance(tc, dict) else None
|
||||
args = fn.get("arguments") if isinstance(fn, dict) else None
|
||||
if isinstance(args, str) and int(len(args) * 0.3) > per_call:
|
||||
if isinstance(args, str) and estimate_text_tokens(args) > per_call:
|
||||
new_fn = dict(fn)
|
||||
new_fn["arguments"] = json.dumps({"_truncated_for_context": len(args)})
|
||||
new_tc = dict(tc)
|
||||
@@ -260,10 +340,29 @@ def trim_for_context(messages: List[Dict], context_length: int, reserve_tokens:
|
||||
# metadata as essential alongside the leading system prompt.
|
||||
def _is_research_primer(m):
|
||||
return bool((m.get("metadata") or {}).get("research_spinoff_from"))
|
||||
|
||||
def _is_compaction_summary(m):
|
||||
content = m.get("content")
|
||||
return isinstance(content, str) and content.lstrip().startswith(
|
||||
("[Conversation summary", "Conversation Summary")
|
||||
)
|
||||
|
||||
_primers = [m for m in system_msgs if _is_research_primer(m)]
|
||||
_non_primer = [m for m in system_msgs if not _is_research_primer(m)]
|
||||
essential_system = (_non_primer[:1] if _non_primer else []) + _primers
|
||||
extra_system = _non_primer[1:]
|
||||
_summaries = [m for m in system_msgs if _is_compaction_summary(m)]
|
||||
_non_essential = [
|
||||
m for m in system_msgs
|
||||
if not _is_research_primer(m) and not _is_compaction_summary(m)
|
||||
]
|
||||
# The base prompt and the conversation summary are the minimum state
|
||||
# needed to continue a task. Summaries used to be classified as ordinary
|
||||
# extra system context, so a large route prompt could trim them away right
|
||||
# after compaction and send the model the same lost-context request again.
|
||||
essential_system = (
|
||||
(_non_essential[:1] if _non_essential else [])
|
||||
+ _primers
|
||||
+ _summaries
|
||||
)
|
||||
extra_system = _non_essential[1:]
|
||||
|
||||
# Try dropping extra system messages one by one (from the end)
|
||||
trimmed = essential_system + convo_msgs
|
||||
@@ -289,31 +388,82 @@ def trim_for_context(messages: List[Dict], context_length: int, reserve_tokens:
|
||||
if estimate_tokens(trimmed) <= budget:
|
||||
return _sanitize_tool_messages(essential_system + protected_msgs + convo_msgs)
|
||||
|
||||
# Still too big — drop older conversation turns BUT always keep the current
|
||||
# user turn. If a pasted message alone exceeds the model context, truncate
|
||||
# that message with a visible notice instead of dropping it; otherwise the
|
||||
# model appears to "ignore" large pastes because it never receives them.
|
||||
# Hermes-style: recent context matters more than old context.
|
||||
# Still too big — drop older conversation turns BUT always keep the latest
|
||||
# user turn. After a tool round the final message is usually an assistant
|
||||
# tool-call or a tool result, not the user's request; treating the last
|
||||
# message as "current" drops the real question and lets the model answer
|
||||
# stale context on the follow-up round.
|
||||
PROTECT_RECENT = 10
|
||||
current_msg = convo_msgs[-1:] if convo_msgs else []
|
||||
prior_convo = convo_msgs[:-1] if convo_msgs else []
|
||||
if len(prior_convo) >= PROTECT_RECENT:
|
||||
old_msgs = prior_convo[:-(PROTECT_RECENT - 1)]
|
||||
recent_msgs = prior_convo[-(PROTECT_RECENT - 1):] + current_msg
|
||||
def _is_direct_user_message(message: Dict) -> bool:
|
||||
# Qwen transports runtime corrections as user-role messages. Their
|
||||
# server-owned provenance must not make them replace the real request
|
||||
# as the protected tail's anchor. Never infer this from prompt wording.
|
||||
if message.get("role") != "user" or message.get("_harness_control"):
|
||||
return False
|
||||
metadata = message.get("metadata") or {}
|
||||
# Textual tool transports intentionally wrap external results as user
|
||||
# messages. They are context for the request, not a new request.
|
||||
return not (
|
||||
metadata.get("trusted") is False
|
||||
and bool(metadata.get("source"))
|
||||
)
|
||||
|
||||
latest_user_idx = -1
|
||||
for idx in range(len(convo_msgs) - 1, -1, -1):
|
||||
if _is_direct_user_message(convo_msgs[idx]):
|
||||
latest_user_idx = idx
|
||||
break
|
||||
if latest_user_idx < 0:
|
||||
# Preserve the historical fallback for callers that only supply
|
||||
# synthetic context and no direct user turn.
|
||||
for idx in range(len(convo_msgs) - 1, -1, -1):
|
||||
if convo_msgs[idx].get("role") == "user":
|
||||
latest_user_idx = idx
|
||||
break
|
||||
if latest_user_idx >= 0:
|
||||
current_tail = convo_msgs[latest_user_idx:]
|
||||
prior_convo = convo_msgs[:latest_user_idx]
|
||||
else:
|
||||
current_tail = convo_msgs[-1:] if convo_msgs else []
|
||||
prior_convo = convo_msgs[:-1] if convo_msgs else []
|
||||
|
||||
recent_prior_count = max(0, PROTECT_RECENT - len(current_tail))
|
||||
if len(prior_convo) > recent_prior_count:
|
||||
old_msgs = prior_convo[:-recent_prior_count] if recent_prior_count else prior_convo[:]
|
||||
recent_msgs = (prior_convo[-recent_prior_count:] if recent_prior_count else []) + current_tail
|
||||
while old_msgs and estimate_tokens(essential_system + old_msgs + recent_msgs) > budget:
|
||||
old_msgs.pop(0)
|
||||
convo_msgs = old_msgs + recent_msgs
|
||||
else:
|
||||
convo_msgs = prior_convo + current_msg
|
||||
while prior_convo and estimate_tokens(essential_system + prior_convo + current_msg) > budget:
|
||||
while prior_convo and estimate_tokens(essential_system + prior_convo + current_tail) > budget:
|
||||
prior_convo.pop(0)
|
||||
convo_msgs = prior_convo + current_msg
|
||||
convo_msgs = prior_convo + current_tail
|
||||
|
||||
# If the current message itself is too large, shrink only that message.
|
||||
if current_msg and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
|
||||
prefix = essential_system + protected_msgs + convo_msgs[:-1]
|
||||
# If the current request + tool tail is still too large, shrink that tail
|
||||
# instead of dropping the latest user turn. Native tool responses can be
|
||||
# huge (web search/fetch), and preserving a truncated source block is better
|
||||
# than sending the model a prompt with no active user request.
|
||||
if current_tail and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
|
||||
tail_start = len(convo_msgs) - len(current_tail)
|
||||
prefix = essential_system + protected_msgs + convo_msgs[:tail_start]
|
||||
available_for_tail = max(64 * len(current_tail), budget - estimate_tokens(prefix))
|
||||
per_tail_msg = max(64, available_for_tail // max(1, len(current_tail)))
|
||||
convo_msgs[tail_start:] = [
|
||||
_truncate_message_to_token_budget(msg, per_tail_msg)
|
||||
for msg in current_tail
|
||||
]
|
||||
|
||||
# Last ditch: if the tool tail still cannot fit, keep only the latest user
|
||||
# message (truncated if needed). Losing source output is bad; losing the user
|
||||
# request is worse and caused visibly crossed answers.
|
||||
if current_tail and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
|
||||
latest_user = next(
|
||||
(m for m in current_tail if _is_direct_user_message(m)),
|
||||
next((m for m in current_tail if m.get("role") == "user"), current_tail[0]),
|
||||
)
|
||||
prefix = essential_system + protected_msgs
|
||||
available_for_current = max(64, budget - estimate_tokens(prefix))
|
||||
convo_msgs[-1] = _truncate_message_to_token_budget(convo_msgs[-1], available_for_current)
|
||||
convo_msgs = [_truncate_message_to_token_budget(latest_user, available_for_current)]
|
||||
|
||||
result = _sanitize_tool_messages(essential_system + protected_msgs + convo_msgs)
|
||||
logger.info(f"Trimmed to {estimate_tokens(result)} tokens ({len(result)} messages)")
|
||||
@@ -330,6 +480,7 @@ async def maybe_compact(
|
||||
*,
|
||||
persist: bool = True,
|
||||
compaction_state: Optional[Dict[str, Any]] = None,
|
||||
deterministic: bool = False,
|
||||
) -> tuple:
|
||||
"""Check context usage and compact if above threshold.
|
||||
|
||||
@@ -338,14 +489,38 @@ async def maybe_compact(
|
||||
context_length = get_context_length(endpoint_url, model)
|
||||
used = estimate_tokens(messages)
|
||||
pct = (used / context_length) * 100 if context_length else 0
|
||||
threshold = auto_compact_threshold_percent()
|
||||
|
||||
if pct < COMPACT_THRESHOLD * 100:
|
||||
if pct < threshold:
|
||||
return messages, context_length, False
|
||||
|
||||
logger.info(
|
||||
f"Context at {pct:.1f}% ({used}/{context_length} tokens) — compacting"
|
||||
f"Context at {pct:.1f}% ({used}/{context_length} tokens, threshold={threshold}%) — compacting"
|
||||
)
|
||||
|
||||
if deterministic:
|
||||
# Unattended workers must not compete with policy generation for the
|
||||
# same saturated endpoint just to summarize their own transcript. Trim
|
||||
# to 75% so the next tool round has useful headroom. This path is not
|
||||
# persisted into interactive history and preserves the active request
|
||||
# through trim_for_context's protected-tail rules.
|
||||
reserve_tokens = max(512, int(context_length * 0.25))
|
||||
compacted = trim_for_context(
|
||||
messages,
|
||||
context_length,
|
||||
reserve_tokens=reserve_tokens,
|
||||
)
|
||||
changed = compacted != messages
|
||||
if changed:
|
||||
logger.info(
|
||||
"Deterministically compacted: %s -> %s tokens (%s -> %s messages)",
|
||||
used,
|
||||
estimate_tokens(compacted),
|
||||
len(messages),
|
||||
len(compacted),
|
||||
)
|
||||
return compacted, context_length, changed
|
||||
|
||||
# Split into system preface and conversation
|
||||
system_msgs = []
|
||||
convo_msgs = []
|
||||
|
||||
Reference in New Issue
Block a user