mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-08 07:52:20 +02:00
fix(effects): require evidence for external completion claims
Effect obligations were consulted only for declared artifacts, and reported
external success could be presented as done. Now, regardless of declared
artifacts:
- the latest effect on any changed file contradicted by a fresh readback
fails the run (a superseded earlier effect is history, not a contradiction);
- a passing verifier followed by an effect that may have changed state
without settled evidence is stale (BLOCKED);
- executed external effects that are not VERIFIED cap the decision at
UNVERIFIED, and the answer always carries server-authored facts for them
("reported success; any external change it made was not independently
verified", "reported failure", "unknown outcome").
The disclosure is structural and does not depend on recognizing the model's
wording. When it is the only change, the model's answer events are released
unchanged and the disclosure follows as one delta (and in round_texts).
Prose filtering is also tightened (remote verbs are mutation claims, an
unnamed "I updated it" cannot borrow the single required artifact, bare
"Done." is a terminal claim beside unverified external effects). A passing
verifier still supports test claims; it never speaks for the external effect.
Replaces the uncommitted attempt that blocked every run with any RUNNING
effect: a background launch with no declared obligations completes
UNVERIFIED.
This commit is contained in:
@@ -42,9 +42,10 @@ _TEST_STATUS_CLAIM = re.compile(
|
||||
r'(?:pass(?:ed|ing)?|succeeded|successful(?:ly)?|green)\b|'
|
||||
r'\b(?:zero|no|0)\s+(?:test\s+)?failures\b', re.I)
|
||||
_EXECUTION_CLAIM = re.compile(
|
||||
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed)|'
|
||||
rf'(?:file|artifact|command|script|service|server|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started)|'
|
||||
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed))\b', re.I)
|
||||
r'\b(?:(?:I|we|I\'ve|we\'ve|and)\s+(?:have\s+)?(?:successfully\s+)?(?:ran|executed|tested|verified|created|updated|modified|wrote|saved|fixed|completed|sent|deleted|submitted|published|deployed|configured|uploaded)|'
|
||||
rf'(?:file|artifact|command|script|service|server|email|message|record|resource|{_ARTIFACT_PATH})\s+(?:was\s+|has\s+been\s+|is\s+)?(?:successfully\s+)?(?:created|updated|written|saved|executed|started|sent|deleted|submitted|published|deployed|configured)|'
|
||||
r'(?:successfully\s+)(?:ran|executed|created|updated|saved|completed|sent|deleted|submitted|published|deployed)|'
|
||||
r'(?:the\s+)?(?:remote\s+)?(?:operation|request|call|mutation|action)\s+(?:was\s+|has\s+)?(?:successfully\s+)?(?:completed|succeeded|finished))\b', re.I)
|
||||
_UNATTESTED_TEST_METRIC = re.compile(
|
||||
r'\b\d+\s+(?:(?:unit|integration)\s+)?tests?\s+pass(?:ed|ing)?\b|'
|
||||
r'\b\d+\s+passed\b|\b\d+(?:\.\d+)?%\s+(?:test\s+)?coverage\b', re.I)
|
||||
@@ -52,7 +53,7 @@ _UNBOUNDED_SUCCESS = re.compile(
|
||||
r'\b(?:everything|all\s+(?:bugs|issues))\s+(?:is\s+|are\s+|has\s+been\s+)?'
|
||||
r'(?:fixed|resolved|working)\b', re.I)
|
||||
_MUTATION_CLAIM = re.compile(
|
||||
r'\b(?:created|updated|modified|wrote|written|saved|fixed)\b', re.I)
|
||||
r'\b(?:created|updated|modified|wrote|written|saved|fixed|sent|deleted|submitted|published|deployed|configured|uploaded)\b', re.I)
|
||||
_TEST_IDENTITY = re.compile(r'\b(?:pytest|unittest)\b', re.I)
|
||||
_TEST_SUBJECT = re.compile(r'\b(?:tests?|test suite|pytest|unittest|checks?|verification)\b', re.I)
|
||||
_CLAIM_PATH = re.compile(_ARTIFACT_PATH)
|
||||
@@ -106,17 +107,20 @@ def _supported_prose(text: str, ledger: EvidenceLedger, decision: CompletionDeci
|
||||
"""Remove unsupported assertions at statement boundaries; add no notice."""
|
||||
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
|
||||
execution_required = _execution_obligation(ledger.requirements)
|
||||
# Bare "Done." cannot stand for an external effect nobody verified.
|
||||
terminal_claims = execution_required or bool(ledger.unverified_external_effects())
|
||||
kept = []
|
||||
removed = ''
|
||||
for statement, scoped in _unquoted_statements(text):
|
||||
why = ''
|
||||
for claim, scope in _current_run_claims(scoped, execution_required=execution_required):
|
||||
for claim, scope in _current_run_claims(scoped, execution_required=terminal_claims):
|
||||
paths = tuple(match.group().rstrip('.') for match in _CLAIM_PATH.finditer(scope))
|
||||
if claim == 'metric':
|
||||
why = 'test counts, coverage or exhaustive correctness were not established by execution evidence'
|
||||
elif claim == 'test':
|
||||
identities = tuple(match.group().lower() for match in _TEST_IDENTITY.finditer(scope))
|
||||
if decision.status != CompletionStatus.VERIFIED or not ledger._supports_verifier_claim(identities, paths):
|
||||
if (decision.status not in {CompletionStatus.VERIFIED, CompletionStatus.UNVERIFIED}
|
||||
or not ledger._supports_verifier_claim(identities, paths)):
|
||||
why = 'no current passing executable verification supports the claim'
|
||||
elif claim == 'mutation':
|
||||
if not ledger._supports_artifact_claim(EvidenceKind.ARTIFACT_MUTATION, paths):
|
||||
@@ -142,7 +146,22 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec
|
||||
Exit status proves neither test counts nor coverage. A bad assertion is
|
||||
removed at statement boundaries instead of erasing an entire explanation.
|
||||
The execution outcome remains separate from a discarded model assertion.
|
||||
Unverified external effects are always stated by the server, so no
|
||||
surviving prose can present a reported remote success as a verified one.
|
||||
"""
|
||||
answer, reason = _completion_answer(text, ledger, decision)
|
||||
return _disclose(answer, ledger), reason
|
||||
|
||||
|
||||
def _disclose(answer: str, ledger: EvidenceLedger) -> str:
|
||||
"""Append the server's facts for unverified external effects."""
|
||||
summary = ' '.join(ledger.effect_disclosures())
|
||||
if not summary:
|
||||
return answer
|
||||
return (answer.rstrip() + '\n\n' + summary) if answer.strip() else summary
|
||||
|
||||
|
||||
def _completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDecision) -> tuple[str, str]:
|
||||
incomplete = decision.reason if not decision.can_complete and decision.status != CompletionStatus.AWAITING_USER else ''
|
||||
execution_required = _execution_obligation(ledger.requirements)
|
||||
prose, removed = _supported_prose(text, ledger, decision)
|
||||
@@ -163,7 +182,7 @@ def completion_answer(text: str, ledger: EvidenceLedger, decision: CompletionDec
|
||||
facts = []
|
||||
if ledger.requirements.required_artifacts:
|
||||
facts.append('Output available: ' + ', '.join(ledger.requirements.required_artifacts) + '.')
|
||||
if decision.status == CompletionStatus.VERIFIED:
|
||||
if decision.status == CompletionStatus.VERIFIED or ledger._supports_verifier_claim():
|
||||
facts.append('The latest executable verification passed.')
|
||||
elif any(e.kind == EvidenceKind.ARTIFACT_VALIDATION and e.authoritative and e.success for e in ledger.events):
|
||||
facts.append('Artifact readback verified. No passing executable test result was recorded.')
|
||||
@@ -312,7 +331,8 @@ def with_completion_gate(func):
|
||||
# Exhaustion limits execution; factual source synthesis can remain
|
||||
# useful and must not be replaced merely because the budget ended.
|
||||
presentation_decision = ledger.evaluate(awaiting_user=awaiting) if exhausted and not provider_error else decision
|
||||
safe_answer, reason = completion_answer(answer, ledger, presentation_decision)
|
||||
filtered_answer, reason = _completion_answer(answer, ledger, presentation_decision)
|
||||
safe_answer = _disclose(filtered_answer, ledger)
|
||||
# Evaluate each earlier draft as well as the final replacement.
|
||||
# Never replay an unsupported intermediate success claim.
|
||||
draft = ''.join(str(e.get('delta') or e.get('content') or '')
|
||||
@@ -328,7 +348,14 @@ def with_completion_gate(func):
|
||||
released_at = perf_counter()
|
||||
if not provider_error:
|
||||
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
|
||||
replaced_answer = bool(presentation_replaced or reason or unsafe_draft or safe_answer != answer)
|
||||
# When the only change is the server's effect disclosure, the
|
||||
# model's answer events are released unchanged and the disclosure
|
||||
# follows them, so no earlier-round text is dropped.
|
||||
disclosure = safe_answer[len(filtered_answer):] if safe_answer != filtered_answer else ''
|
||||
disclosure_only = bool(disclosure) and not (presentation_replaced or reason or unsafe_draft
|
||||
or filtered_answer != answer)
|
||||
replaced_answer = not disclosure_only and bool(
|
||||
presentation_replaced or reason or unsafe_draft or safe_answer != answer)
|
||||
if replaced_answer:
|
||||
reasoning = [event for event in answer_events if event.get('thinking') is True]
|
||||
_, unsafe_reasoning = completion_answer(
|
||||
@@ -341,6 +368,8 @@ def with_completion_gate(func):
|
||||
else:
|
||||
for event in answer_events:
|
||||
yield _event(event)
|
||||
if disclosure_only:
|
||||
yield _event({'delta': disclosure})
|
||||
if provider_error:
|
||||
yield _event({'type': 'completion_decision', 'data': decision.to_dict()})
|
||||
for event in metrics_events:
|
||||
@@ -360,6 +389,10 @@ def with_completion_gate(func):
|
||||
if not provider_error:
|
||||
metadata['round_texts'] = [safe_answer]
|
||||
metadata['completion_gate_reason'] = reason or unsafe_draft or 'receipt_summary'
|
||||
elif disclosure_only and metadata.get('round_texts') and isinstance(metadata['round_texts'], list) \
|
||||
and isinstance(metadata['round_texts'][-1], str):
|
||||
# Reload renders round_texts: keep the disclosure with them.
|
||||
metadata['round_texts'] = [*metadata['round_texts'][:-1], metadata['round_texts'][-1] + disclosure]
|
||||
if provider_error and isinstance(metadata.get('round_texts'), list):
|
||||
# Failed rounds stay as per-round diagnostics, but they are
|
||||
# rendered again on reload. Apply the same statement filter
|
||||
|
||||
@@ -90,7 +90,7 @@ class ActionJournal:
|
||||
outcome = history.latest_outcome(assessment.effect_id)
|
||||
entries.append({
|
||||
'ordinal': order.get(assessment.action_id), 'assessment': assessment,
|
||||
'tool': claim.operation.tool, 'unknown_scope': claim.unknown_scope,
|
||||
'tool': claim.operation.tool, 'unknown_scope': claim.unknown_scope, 'external': claim.external,
|
||||
'paths': tuple(ref.location[-1] for ref in claim.impact_scope if ref.kind.value == 'filesystem'),
|
||||
'mutation_attempted': bool(outcome and outcome.facts.mutation_attempted),
|
||||
'artifact_changes': changes.get(assessment.action_id),
|
||||
|
||||
Reference in New Issue
Block a user