From a23d709056f0949963c467e0df49e117aab4770c Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 21:28:52 +0000 Subject: [PATCH] Record controlled tool-choice argument failures and broad replay --- docs/search-quality-audit-20260917.md | 4 ++++ scripts/probe_search_tool_choice.mjs | 23 +++++++++++++++++++++++ 2 files changed, 27 insertions(+) create mode 100644 scripts/probe_search_tool_choice.mjs diff --git a/docs/search-quality-audit-20260917.md b/docs/search-quality-audit-20260917.md index f8e122323..8fdc4cea2 100644 --- a/docs/search-quality-audit-20260917.md +++ b/docs/search-quality-audit-20260917.md @@ -75,6 +75,10 @@ Further provider inspection found that the news-to-general fallback dropped the ### Additional informal/multi-part live checks +Full 23-conversation regression launched on `6000b718`/current deployed harness: `reports/clean-v3-search-quality-2026-09-17T21-27-27-686Z.json`. Active handle recorded in session; do not restart based on elapsed observation time. + +Read-only tool-choice control `reports/search-tool-choice-probe-1789680509169.json` uses the exact compact web_search schema, a short system prompt, identical user prompts/temperature/model, and never executes emitted calls. For both weekly-news and typo-news prompts, auto/required emitted nonempty queries. Forced named mode emitted an extraneous `command` field in both; typo-news omitted query entirely. Six calls are preliminary evidence of tool-choice/schema behavior, not proof of a universal backend defect or a production fix. Next test should use the canonical harness system/history before changing dispatch. Probe script saves full schemas and public emitted calls for reproducibility. + `reports/search-synthesis-probe-1789680321559.json` compares identical saved native tool history with/without `_harness_control` messages, same canonical base prompt, no offered tools. Full trace: short answer without links, 2.59s. Controls removed: longer answer with links, 6.39s, but introduced a Do Not Track URL not established by the recorded evidence. This is not grounds to remove recovery controls wholesale or claim a factual quality win. Weekly-news replay `reports/clean-v3-search-quality-2026-09-17T21-24-11-361Z.json` corrected the missing query but still returned no evidence (16.35s). Direct simultaneous provider control with exact query `AI developments this week`, `time_filter=week`: general returned zero, news five. `3ea5a348` recognizes time-qualified developments as news intent while retaining general routing for tutorials, historical discussion, software versions and documentation. Provider/publication/query-relaxation tests: 80 passed. Live weekly-news replay launched after deployment; returned results still require relevance/source review. diff --git a/scripts/probe_search_tool_choice.mjs b/scripts/probe_search_tool_choice.mjs new file mode 100644 index 000000000..2e1084760 --- /dev/null +++ b/scripts/probe_search_tool_choice.mjs @@ -0,0 +1,23 @@ +#!/usr/bin/env node +// Read-only endpoint probe: emitted tool calls are recorded, never executed. +import fs from 'node:fs'; +import {execFileSync} from 'node:child_process'; +const tools = JSON.parse(execFileSync((process.env.PYTHON || "python3"), ['-c', + 'import json; from src.clean_agent_preview import compact_schemas; from src.tool_schemas import FUNCTION_TOOL_SCHEMAS; print(json.dumps(compact_schemas([s for s in FUNCTION_TOOL_SCHEMAS if s["function"]["name"] == "web_search"])))'], {encoding:'utf8'})); +const model = process.env.MODEL || 'model-f'; +const endpoint = process.env.ENDPOINT_URL || (() => { throw new Error("ENDPOINT_URL is required"); })(); +const prompts = ['Catch me up on the biggest AI developments this week. Explain why they matter and link your sources.', 'serch latest ai news pls']; +const results = []; +for (const prompt of prompts) { + for (const choice of ['auto', 'required', {type:'function', function:{name:'web_search'}}]) { + const started = performance.now(); + const response = await fetch(endpoint, {method:'POST', headers:{'Content-Type':'application/json'}, signal:AbortSignal.timeout(90000), + body:JSON.stringify({model, messages:[{role:'system',content:'You are Odysseus. Use web_search to find current information relevant to the user request.'},{role:'user',content:prompt}], tools, tool_choice:choice, temperature:0, max_tokens:256, stream:false, chat_template_kwargs:{enable_thinking:false}})}); + const data = await response.json(); + const result = {prompt,choice,status:response.status,seconds:(performance.now()-started)/1000,message:data.choices?.[0]?.message,error:data.error}; + results.push(result); console.log(JSON.stringify(result)); + } +} +const target = `reports/search-tool-choice-probe-${Date.now()}.json`; +fs.writeFileSync(target, JSON.stringify({model,tools,results},null,2)+'\n'); +console.log(target);