Files
odysseus/tests/smoke/test_compare_smoke.py
T
Léo 12f74ec9ea test(smoke): add a release smoke suite over every advertised feature area
The decomposition lanes have two safety nets and neither covers the
product. The checkpoint benchmark measures the agent runtime; the
computed-style snapshot pins the CSS. Nothing checked that Notes,
Calendar, Documents, Email, Memory, Cookbook or Settings still worked
after a route package moved or a 17,000-line module was split - and the
unit suite does not, since a byte-identical file move can break tests
that pass on the base branch with CI green throughout. The 28 Playwright
specs we do have are all under tests/e2e/photo-editor/ and no workflow
runs them.

scripts/odysseus-smoke boots this worktree through `odysseus dev` and
runs tests/smoke/: one scenario per area, each asserting a user-visible
outcome rather than a status code. Models come from a deterministic
OpenAI-compatible stub on an ephemeral loopback port; email reuses the
existing ODYSSEUS_EMAIL_FIXTURE path rather than inventing a second
mechanism. No scenario touches a live endpoint or the network.

The report is a per-area table that prints the areas the suite does not
cover next to the ones it does, and builds its rows from the registry
rather than from what happened to run, so an area cannot go missing by
having its module deleted or renamed. Under a plain pytest with nothing
booted every scenario skips with the reason, so the full suite stays
green.
2026-09-30 12:12:58 +02:00

72 lines
2.7 KiB
Python

"""Compare: a blind comparison streams both sides and reveals them on the vote.
Two model ids on the one stub provider is what makes this checkable
without a second endpoint: each returns a reply naming itself, so the
reveal can be matched against which text arrived on which side.
"""
from __future__ import annotations
import json
from tests.smoke.stub_provider import MODEL_PRIMARY, MODEL_SECONDARY, reply_for
COMPARE_PATH = "/api/compare"
CHAT_STREAM_PATH = "/api/chat_stream"
PROMPT = "Smoke check: compare two replies."
def _stream_text(client, session_id: str) -> str:
deltas = []
with client.stream("POST", CHAT_STREAM_PATH,
json={"message": PROMPT, "session": session_id}) as response:
assert response.status_code == 200
for line in response.iter_lines():
if not line.startswith("data: "):
continue
payload = line[len("data: "):].strip()
if payload == "[DONE]":
break
try:
event = json.loads(payload)
except ValueError:
continue
if "delta" in event:
deltas.append(str(event["delta"]))
return "".join(deltas)
def test_a_blind_comparison_streams_and_reveals(client, stub_endpoint):
started = client.post(f"{COMPARE_PATH}/start", data={
"prompt": PROMPT,
"model_a": MODEL_PRIMARY,
"model_b": MODEL_SECONDARY,
"endpoint_a_id": stub_endpoint,
"endpoint_b_id": stub_endpoint,
"is_blind": "true",
})
assert started.status_code == 200, started.text
comparison = started.json()
comparison_id = comparison["id"]
# Blind: the start response must not say which model is on which side.
assert not comparison.get("model_left"), comparison
assert not comparison.get("model_right"), comparison
left = _stream_text(client, comparison["session_left"])
right = _stream_text(client, comparison["session_right"])
assert {left, right} == {reply_for(MODEL_PRIMARY), reply_for(MODEL_SECONDARY)}, (left, right)
voted = client.post(f"{COMPARE_PATH}/{comparison_id}/vote", data={"winner": "left"})
assert voted.status_code == 200, voted.text
revealed = voted.json().get("revealed") or {}
assert revealed.get("left") in (MODEL_PRIMARY, MODEL_SECONDARY), voted.text
assert reply_for(revealed["left"]) == left, (revealed, left)
assert reply_for(revealed["right"]) == right, (revealed, right)
history = client.get(f"{COMPARE_PATH}/history")
assert history.status_code == 200, history.text
entries = [row for row in history.json() if row.get("id") == comparison_id]
assert entries, history.text
assert entries[0].get("winner"), entries[0]