refactor(bench): share the GPU residency guard, and guard tool calling too
Extracts the residency snapshot/restore into scripts/ollama_residency.py so the two benchmarks cannot drift, and applies it to benchmark_tool_calling.py, which had no protection at all. That script was the more dangerous of the two. It rewrites OLLAMA_DEFAULT_MODEL in .env and lets uvicorn reload onto it, restoring the original only after the loop — so any crash or interrupt left the *running server* pointed at the benchmark model. Its DEFAULT_MODELS begins with mistral-nemo-large, the 9.2G model implicated in the 2026-08-07 VRAM outage. Both the .env restore and the residency restore now run from `finally`. SIGTERM is handled explicitly in the shared module. Python runs `finally` for SIGINT, which arrives as KeyboardInterrupt, but the default SIGTERM action terminates outright, so `timeout` or a plain `kill` skipped the guard entirely. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -17,12 +17,16 @@ import asyncio
|
||||
import json
|
||||
import re
|
||||
import statistics
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
from scripts.ollama_residency import install_sigterm_handler, residency_guard
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Configuration
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -525,18 +529,31 @@ async def main():
|
||||
original_env = ENV_PATH.read_text()
|
||||
|
||||
all_stats = []
|
||||
async with httpx.AsyncClient() as client:
|
||||
for model in models:
|
||||
stats = await benchmark_model(client, model, args.iterations)
|
||||
all_stats.append(stats)
|
||||
|
||||
# Restore original .env
|
||||
ENV_PATH.write_text(original_env)
|
||||
print(f"\n .env restored to original")
|
||||
# Both restores must survive a crash or an interrupt. The .env one especially:
|
||||
# this script rewrites OLLAMA_DEFAULT_MODEL and lets uvicorn reload onto it,
|
||||
# so bailing out mid-run used to leave the *running server* pointed at the
|
||||
# benchmark model — and DEFAULT_MODELS starts at mistral-nemo-large, the 9.2G
|
||||
# model implicated in the 2026-08-07 VRAM outage.
|
||||
install_sigterm_handler()
|
||||
try:
|
||||
with residency_guard(models_used=models):
|
||||
async with httpx.AsyncClient() as client:
|
||||
for model in models:
|
||||
stats = await benchmark_model(client, model, args.iterations)
|
||||
all_stats.append(stats)
|
||||
finally:
|
||||
ENV_PATH.write_text(original_env)
|
||||
print("\n .env restored to original")
|
||||
|
||||
print_comparison(all_stats)
|
||||
save_results(all_stats, Path(args.output))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
try:
|
||||
asyncio.run(main())
|
||||
except KeyboardInterrupt:
|
||||
# .env and GPU residency are both restored by now; do not bury that
|
||||
# output under a traceback.
|
||||
print("\ninterrupted", file=sys.stderr)
|
||||
raise SystemExit(130) from None
|
||||
|
||||
Reference in New Issue
Block a user