Squash Odysseus development history

This commit is contained in:
pewdiepie-archdaemon
2026-09-11 06:04:19 +00:00
parent c9dd68d890
commit 84aa9a91de
871 changed files with 265870 additions and 27854 deletions
+207 -32
View File
@@ -10,9 +10,10 @@ import logging
import re
from typing import Any, Dict, List, Optional
from src.model_context import get_context_length, estimate_tokens
from src.model_context import estimate_text_tokens, get_context_length, estimate_tokens
from src.llm_core import llm_call_async
from src.endpoint_resolver import resolve_endpoint
from src.settings import get_setting
from core.models import ChatMessage
logger = logging.getLogger(__name__)
@@ -37,10 +38,72 @@ def _content_as_text(content: Any) -> str:
return ""
_MULTIMODAL_IMAGE_TYPES = {"image_url", "input_image", "image"}
def prune_multimodal_images(
messages: List[Dict],
*,
max_images: int = 8,
) -> List[Dict]:
"""Keep uniformly sampled visual blocks across a multimodal history."""
image_locations = []
for message_index, message in enumerate(messages):
content = message.get("content") if isinstance(message, dict) else None
if not isinstance(content, list):
continue
for content_index, item in enumerate(content):
if isinstance(item, dict) and item.get("type") in _MULTIMODAL_IMAGE_TYPES:
image_locations.append((message_index, content_index))
limit = max(0, int(max_images))
if len(image_locations) <= limit:
return list(messages)
if limit == 0:
selected = set()
elif limit == 1:
selected = {image_locations[-1]}
else:
last = len(image_locations) - 1
selected = {
image_locations[round(index * last / (limit - 1))]
for index in range(limit)
}
pruned = []
for message_index, message in enumerate(messages):
if not isinstance(message, dict):
pruned.append(message)
continue
cloned = dict(message)
content = message.get("content")
if isinstance(content, list):
cloned["content"] = [
item for content_index, item in enumerate(content)
if not (
isinstance(item, dict)
and item.get("type") in _MULTIMODAL_IMAGE_TYPES
and (message_index, content_index) not in selected
)
]
pruned.append(cloned)
return pruned
COMPACT_THRESHOLD = 0.85 # Trigger compaction at 85% of context window
SUMMARY_MAX_TOKENS = 1024
SMALL_CONTEXT_LIMIT = 8192 # Models with context <= this get aggressive trimming
def auto_compact_threshold_percent() -> int:
"""Configured auto-compaction threshold, clamped to a sane UI range."""
try:
value = int(get_setting("auto_compact_threshold_percent", int(COMPACT_THRESHOLD * 100)) or 85)
except (TypeError, ValueError):
value = int(COMPACT_THRESHOLD * 100)
return max(50, min(95, value))
# Cursor-style self-summarization prompt — produces structured, dense summaries
SELF_SUMMARY_SYSTEM_PROMPT = """You are summarizing a conversation to preserve context after compaction. Produce a structured summary that lets the conversation continue seamlessly.
@@ -128,7 +191,7 @@ def _sanitize_tool_messages(msgs: List[Dict]) -> List[Dict]:
def _message_text_token_estimate(text: str) -> int:
if not isinstance(text, str):
return 4
return int(len(text) * 0.3) + 4
return estimate_text_tokens(text) + 4
def _truncate_text_to_token_budget(text: str, token_budget: int) -> str:
@@ -141,19 +204,36 @@ def _truncate_text_to_token_budget(text: str, token_budget: int) -> str:
# string rather than the raw non-string (which would move the crash
# into the caller that concatenates/measures the result).
return ""
# Match src.model_context.estimate_tokens' rough chars * 0.3 estimate.
max_chars = max(200, int((token_budget - 16) / 0.3))
if len(text) <= max_chars:
if estimate_text_tokens(text) <= token_budget - 16:
return text
notice = (
"\n\n[Notice: the pasted message was too large for this model's context "
"window, so Odysseus kept the beginning and end.]"
)
keep_chars = max(200, max_chars - len(notice))
head_len = max(100, int(keep_chars * 0.7))
tail_len = max(80, keep_chars - head_len)
return text[:head_len].rstrip() + notice + "\n\n" + text[-tail_len:].lstrip()
# Binary replacement characters and dense scripts can approach one token
# per character, while ASCII prose is closer to the historical 0.3 ratio.
# Find the largest head/tail sample that fits the shared estimator instead
# of guessing a character count from one language family.
target = max(64, token_budget - estimate_text_tokens(notice) - 16)
low, high = 1, len(text)
best = 1
while low <= high:
keep_chars = (low + high) // 2
head_len = max(1, int(keep_chars * 0.7))
tail_len = max(0, keep_chars - head_len)
sample = text[:head_len]
if tail_len:
sample += text[-tail_len:]
if estimate_text_tokens(sample) <= target:
best = keep_chars
low = keep_chars + 1
else:
high = keep_chars - 1
head_len = max(1, int(best * 0.7))
tail_len = max(0, best - head_len)
tail = text[-tail_len:].lstrip() if tail_len else ""
return text[:head_len].rstrip() + notice + ("\n\n" + tail if tail else "")
def _truncate_tool_call_args(msg: Dict[str, Any], token_budget: int) -> Dict[str, Any]:
@@ -180,7 +260,7 @@ def _truncate_tool_call_args(msg: Dict[str, Any], token_budget: int) -> Dict[str
for tc in tool_calls:
fn = tc.get("function") if isinstance(tc, dict) else None
args = fn.get("arguments") if isinstance(fn, dict) else None
if isinstance(args, str) and int(len(args) * 0.3) > per_call:
if isinstance(args, str) and estimate_text_tokens(args) > per_call:
new_fn = dict(fn)
new_fn["arguments"] = json.dumps({"_truncated_for_context": len(args)})
new_tc = dict(tc)
@@ -260,10 +340,29 @@ def trim_for_context(messages: List[Dict], context_length: int, reserve_tokens:
# metadata as essential alongside the leading system prompt.
def _is_research_primer(m):
return bool((m.get("metadata") or {}).get("research_spinoff_from"))
def _is_compaction_summary(m):
content = m.get("content")
return isinstance(content, str) and content.lstrip().startswith(
("[Conversation summary", "Conversation Summary")
)
_primers = [m for m in system_msgs if _is_research_primer(m)]
_non_primer = [m for m in system_msgs if not _is_research_primer(m)]
essential_system = (_non_primer[:1] if _non_primer else []) + _primers
extra_system = _non_primer[1:]
_summaries = [m for m in system_msgs if _is_compaction_summary(m)]
_non_essential = [
m for m in system_msgs
if not _is_research_primer(m) and not _is_compaction_summary(m)
]
# The base prompt and the conversation summary are the minimum state
# needed to continue a task. Summaries used to be classified as ordinary
# extra system context, so a large route prompt could trim them away right
# after compaction and send the model the same lost-context request again.
essential_system = (
(_non_essential[:1] if _non_essential else [])
+ _primers
+ _summaries
)
extra_system = _non_essential[1:]
# Try dropping extra system messages one by one (from the end)
trimmed = essential_system + convo_msgs
@@ -289,31 +388,82 @@ def trim_for_context(messages: List[Dict], context_length: int, reserve_tokens:
if estimate_tokens(trimmed) <= budget:
return _sanitize_tool_messages(essential_system + protected_msgs + convo_msgs)
# Still too big — drop older conversation turns BUT always keep the current
# user turn. If a pasted message alone exceeds the model context, truncate
# that message with a visible notice instead of dropping it; otherwise the
# model appears to "ignore" large pastes because it never receives them.
# Hermes-style: recent context matters more than old context.
# Still too big — drop older conversation turns BUT always keep the latest
# user turn. After a tool round the final message is usually an assistant
# tool-call or a tool result, not the user's request; treating the last
# message as "current" drops the real question and lets the model answer
# stale context on the follow-up round.
PROTECT_RECENT = 10
current_msg = convo_msgs[-1:] if convo_msgs else []
prior_convo = convo_msgs[:-1] if convo_msgs else []
if len(prior_convo) >= PROTECT_RECENT:
old_msgs = prior_convo[:-(PROTECT_RECENT - 1)]
recent_msgs = prior_convo[-(PROTECT_RECENT - 1):] + current_msg
def _is_direct_user_message(message: Dict) -> bool:
# Qwen transports runtime corrections as user-role messages. Their
# server-owned provenance must not make them replace the real request
# as the protected tail's anchor. Never infer this from prompt wording.
if message.get("role") != "user" or message.get("_harness_control"):
return False
metadata = message.get("metadata") or {}
# Textual tool transports intentionally wrap external results as user
# messages. They are context for the request, not a new request.
return not (
metadata.get("trusted") is False
and bool(metadata.get("source"))
)
latest_user_idx = -1
for idx in range(len(convo_msgs) - 1, -1, -1):
if _is_direct_user_message(convo_msgs[idx]):
latest_user_idx = idx
break
if latest_user_idx < 0:
# Preserve the historical fallback for callers that only supply
# synthetic context and no direct user turn.
for idx in range(len(convo_msgs) - 1, -1, -1):
if convo_msgs[idx].get("role") == "user":
latest_user_idx = idx
break
if latest_user_idx >= 0:
current_tail = convo_msgs[latest_user_idx:]
prior_convo = convo_msgs[:latest_user_idx]
else:
current_tail = convo_msgs[-1:] if convo_msgs else []
prior_convo = convo_msgs[:-1] if convo_msgs else []
recent_prior_count = max(0, PROTECT_RECENT - len(current_tail))
if len(prior_convo) > recent_prior_count:
old_msgs = prior_convo[:-recent_prior_count] if recent_prior_count else prior_convo[:]
recent_msgs = (prior_convo[-recent_prior_count:] if recent_prior_count else []) + current_tail
while old_msgs and estimate_tokens(essential_system + old_msgs + recent_msgs) > budget:
old_msgs.pop(0)
convo_msgs = old_msgs + recent_msgs
else:
convo_msgs = prior_convo + current_msg
while prior_convo and estimate_tokens(essential_system + prior_convo + current_msg) > budget:
while prior_convo and estimate_tokens(essential_system + prior_convo + current_tail) > budget:
prior_convo.pop(0)
convo_msgs = prior_convo + current_msg
convo_msgs = prior_convo + current_tail
# If the current message itself is too large, shrink only that message.
if current_msg and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
prefix = essential_system + protected_msgs + convo_msgs[:-1]
# If the current request + tool tail is still too large, shrink that tail
# instead of dropping the latest user turn. Native tool responses can be
# huge (web search/fetch), and preserving a truncated source block is better
# than sending the model a prompt with no active user request.
if current_tail and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
tail_start = len(convo_msgs) - len(current_tail)
prefix = essential_system + protected_msgs + convo_msgs[:tail_start]
available_for_tail = max(64 * len(current_tail), budget - estimate_tokens(prefix))
per_tail_msg = max(64, available_for_tail // max(1, len(current_tail)))
convo_msgs[tail_start:] = [
_truncate_message_to_token_budget(msg, per_tail_msg)
for msg in current_tail
]
# Last ditch: if the tool tail still cannot fit, keep only the latest user
# message (truncated if needed). Losing source output is bad; losing the user
# request is worse and caused visibly crossed answers.
if current_tail and estimate_tokens(essential_system + protected_msgs + convo_msgs) > budget:
latest_user = next(
(m for m in current_tail if _is_direct_user_message(m)),
next((m for m in current_tail if m.get("role") == "user"), current_tail[0]),
)
prefix = essential_system + protected_msgs
available_for_current = max(64, budget - estimate_tokens(prefix))
convo_msgs[-1] = _truncate_message_to_token_budget(convo_msgs[-1], available_for_current)
convo_msgs = [_truncate_message_to_token_budget(latest_user, available_for_current)]
result = _sanitize_tool_messages(essential_system + protected_msgs + convo_msgs)
logger.info(f"Trimmed to {estimate_tokens(result)} tokens ({len(result)} messages)")
@@ -330,6 +480,7 @@ async def maybe_compact(
*,
persist: bool = True,
compaction_state: Optional[Dict[str, Any]] = None,
deterministic: bool = False,
) -> tuple:
"""Check context usage and compact if above threshold.
@@ -338,14 +489,38 @@ async def maybe_compact(
context_length = get_context_length(endpoint_url, model)
used = estimate_tokens(messages)
pct = (used / context_length) * 100 if context_length else 0
threshold = auto_compact_threshold_percent()
if pct < COMPACT_THRESHOLD * 100:
if pct < threshold:
return messages, context_length, False
logger.info(
f"Context at {pct:.1f}% ({used}/{context_length} tokens) — compacting"
f"Context at {pct:.1f}% ({used}/{context_length} tokens, threshold={threshold}%) — compacting"
)
if deterministic:
# Unattended workers must not compete with policy generation for the
# same saturated endpoint just to summarize their own transcript. Trim
# to 75% so the next tool round has useful headroom. This path is not
# persisted into interactive history and preserves the active request
# through trim_for_context's protected-tail rules.
reserve_tokens = max(512, int(context_length * 0.25))
compacted = trim_for_context(
messages,
context_length,
reserve_tokens=reserve_tokens,
)
changed = compacted != messages
if changed:
logger.info(
"Deterministically compacted: %s -> %s tokens (%s -> %s messages)",
used,
estimate_tokens(compacted),
len(messages),
len(compacted),
)
return compacted, context_length, changed
# Split into system preface and conversation
system_msgs = []
convo_msgs = []