mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
require breadth for broad current research
This commit is contained in:
@@ -888,19 +888,24 @@ def contentless_final_response(content):
|
||||
))
|
||||
|
||||
|
||||
def incomplete_broad_web_answer(content, user_text):
|
||||
"""Reject a fragmentary answer to a broad current-information request."""
|
||||
def broad_current_web_request(user_text):
|
||||
"""Whether the user requested a broad current-information briefing."""
|
||||
request = str(user_text or '')
|
||||
if not (
|
||||
return bool(
|
||||
re.search(r'\b(?:latest|recent|current|today(?:\'s)?)\b', request, re.I)
|
||||
and re.search(r'\b(?:info(?:rmation)?|news|nees|updates?)\b', request, re.I)
|
||||
):
|
||||
)
|
||||
|
||||
|
||||
def incomplete_broad_web_answer(content, user_text):
|
||||
"""Reject a shallow answer to a broad current-information request."""
|
||||
if not broad_current_web_request(user_text):
|
||||
return False
|
||||
answer = re.sub(r'https?://\S+', ' ', str(content or '')).strip()
|
||||
words = re.findall(r"[A-Za-z0-9][A-Za-z0-9'’-]*", answer)
|
||||
# A broad briefing cannot be fulfilled by one headline fragment. This is
|
||||
# intentionally inapplicable to narrow quick-fact searches.
|
||||
return len(words) < 25
|
||||
return len(words) < 80
|
||||
|
||||
|
||||
def progressive_thinking_for_turn(model, offered_schemas):
|
||||
@@ -4061,6 +4066,28 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
|
||||
})
|
||||
yield event({'type': 'completion_recovery', 'reason': 'contentless_answer'})
|
||||
continue
|
||||
if (
|
||||
broad_current_web_request(direct_user_text)
|
||||
and successful_web_searches == 1
|
||||
and round_number < round_limit
|
||||
):
|
||||
force_web_search_next_round = True
|
||||
replace_streamed_draft_on_finish = True
|
||||
history.pop()
|
||||
history.append({
|
||||
'role': 'user', '_harness_control': True,
|
||||
'content': (
|
||||
'Research breadth check: one search is insufficient for this broad '
|
||||
'current-information request. Run one materially different follow-up '
|
||||
'search that fills gaps or corroborates the strongest findings. Then '
|
||||
'inspect the best source evidence before synthesizing the answer.'
|
||||
),
|
||||
})
|
||||
yield event({
|
||||
'type': 'completion_recovery',
|
||||
'reason': 'insufficient_research_breadth',
|
||||
})
|
||||
continue
|
||||
if (
|
||||
successful_web_searches
|
||||
and incomplete_broad_web_answer(content, direct_user_text)
|
||||
|
||||
@@ -1154,7 +1154,7 @@ async def test_stream_bounds_research_to_two_searches_fetch_then_synthesis(monke
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatch):
|
||||
"""A successful search must not end in a fragmentary headline stub."""
|
||||
"""Broad current research expands, retrieves evidence, then synthesizes."""
|
||||
import src.clean_agent_preview as module
|
||||
|
||||
packets = iter([
|
||||
@@ -1165,10 +1165,30 @@ async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatc
|
||||
},
|
||||
}]}}]},
|
||||
{'choices': [{'delta': {'content': 'Current AI news includes reports about U.'}}]},
|
||||
{'choices': [{'delta': {'tool_calls': [{
|
||||
'index': 0, 'id': 'search-2', 'function': {
|
||||
'name': 'web_search',
|
||||
'arguments': json.dumps({'query': 'AI policy and model releases today'}),
|
||||
},
|
||||
}]}}]},
|
||||
{'choices': [{'delta': {'tool_calls': [{
|
||||
'index': 0, 'id': 'fetch-1', 'function': {
|
||||
'name': 'web_fetch',
|
||||
'arguments': json.dumps({'url': 'https://example.org/ai-news'}),
|
||||
},
|
||||
}]}}]},
|
||||
{'choices': [{'delta': {'content': (
|
||||
'Here is a fuller evidence-based briefing covering the major current AI '
|
||||
'developments, what each source actually reports, and the limits of the '
|
||||
'available evidence. Source: https://example.org/ai-news'
|
||||
'Here is a fuller evidence-based briefing. Recent developments include '
|
||||
'new model releases, updated deployment commitments, and policy proposals '
|
||||
'from several governments. The first source explains what changed in the '
|
||||
'models and how developers can access them. A second independent report '
|
||||
'adds context about evaluation, safety, and likely industry effects. The '
|
||||
'policy coverage distinguishes proposals from rules already in force and '
|
||||
'identifies the dates involved. Taken together, the evidence suggests '
|
||||
'continued rapid deployment alongside stronger demands for transparency, '
|
||||
'although several announced measures remain preliminary. Readers should '
|
||||
'check the linked primary material because this is a changing story. '
|
||||
'Sources: https://example.org/ai-news and https://example.org/ai-policy'
|
||||
)}}]},
|
||||
])
|
||||
requests = []
|
||||
@@ -1191,7 +1211,7 @@ async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatc
|
||||
return Response(next(packets))
|
||||
|
||||
async def execute(block, **kwargs):
|
||||
return 'web_search', {
|
||||
return block.tool_type, {
|
||||
'output': '[1] AI News\n https://example.org/ai-news',
|
||||
'exit_code': 0,
|
||||
'evidence_status': 'available',
|
||||
@@ -1199,23 +1219,30 @@ async def test_stream_retries_an_obviously_truncated_broad_web_answer(monkeypatc
|
||||
|
||||
monkeypatch.setattr(module.httpx, 'AsyncClient', Client)
|
||||
monkeypatch.setattr(module, 'execute_tool_block', execute)
|
||||
schema = next(
|
||||
s for s in FUNCTION_TOOL_SCHEMAS if s['function']['name'] == 'web_search'
|
||||
)
|
||||
contract = resolve_full_inventory_contract(schemas=[schema], policy=ToolPolicy())
|
||||
schemas = [
|
||||
s for s in FUNCTION_TOOL_SCHEMAS
|
||||
if s['function']['name'] in {'web_search', 'web_fetch'}
|
||||
]
|
||||
contract = resolve_full_inventory_contract(schemas=schemas, policy=ToolPolicy())
|
||||
raw = [chunk async for chunk in stream_preview(
|
||||
endpoint_url='http://test', model='test',
|
||||
messages=[{'role': 'user', 'content': 'Latest news in AI?'}],
|
||||
headers={}, turn_contract=contract, session_id='test', owner='test',
|
||||
disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=3,
|
||||
disabled_tools=set(), tool_policy=ToolPolicy(), max_rounds=5,
|
||||
)]
|
||||
events = [json.loads(chunk[6:]) for chunk in raw if '[DONE]' not in chunk]
|
||||
|
||||
assert len(requests) == 3
|
||||
assert 'tools' not in requests[2]
|
||||
assert len(requests) == 5
|
||||
assert requests[2]['tool_choice'] == {
|
||||
'type': 'function', 'function': {'name': 'web_search'},
|
||||
}
|
||||
assert requests[3]['tool_choice'] == {
|
||||
'type': 'function', 'function': {'name': 'web_fetch'},
|
||||
}
|
||||
assert 'tools' not in requests[4]
|
||||
assert any(
|
||||
event.get('type') == 'completion_recovery'
|
||||
and event.get('reason') == 'incomplete_research_answer'
|
||||
and event.get('reason') == 'insufficient_research_breadth'
|
||||
for event in events
|
||||
)
|
||||
assert any(
|
||||
|
||||
Reference in New Issue
Block a user