test(smoke): add a release smoke suite over every advertised feature area

The decomposition lanes have two safety nets and neither covers the
product. The checkpoint benchmark measures the agent runtime; the
computed-style snapshot pins the CSS. Nothing checked that Notes,
Calendar, Documents, Email, Memory, Cookbook or Settings still worked
after a route package moved or a 17,000-line module was split - and the
unit suite does not, since a byte-identical file move can break tests
that pass on the base branch with CI green throughout. The 28 Playwright
specs we do have are all under tests/e2e/photo-editor/ and no workflow
runs them.

scripts/odysseus-smoke boots this worktree through `odysseus dev` and
runs tests/smoke/: one scenario per area, each asserting a user-visible
outcome rather than a status code. Models come from a deterministic
OpenAI-compatible stub on an ephemeral loopback port; email reuses the
existing ODYSSEUS_EMAIL_FIXTURE path rather than inventing a second
mechanism. No scenario touches a live endpoint or the network.

The report is a per-area table that prints the areas the suite does not
cover next to the ones it does, and builds its rows from the registry
rather than from what happened to run, so an area cannot go missing by
having its module deleted or renamed. Under a plain pytest with nothing
booted every scenario skips with the reason, so the full suite stays
green.
This commit is contained in:
Léo
2026-09-30 12:12:58 +02:00
parent 6105702901
commit 12f74ec9ea
19 changed files with 1689 additions and 0 deletions
+209
View File
@@ -0,0 +1,209 @@
#!/usr/bin/env python3
"""odysseus-smoke — boot this worktree and drive every advertised feature area once.
The decomposition work has two safety nets and neither one covers the
product: the checkpoint benchmark measures the agent runtime, and the
computed-style snapshot pins the CSS. Nothing checked that Notes,
Calendar, Documents, Email, Memory, Cookbook or Settings still worked
after a route package moved or a 17,000-line module was split. This is
that check, and it is deliberately shallow: one scenario per area,
asserting a user-visible outcome rather than an HTTP 200.
It owns no instance logic. `odysseus dev` already isolates the ports,
the data dir and ChromaDB per worktree, so this boots through it, hands
the details to pytest in the environment, and stops what it started.
odysseus smoke # boot, run every area, stop again
odysseus smoke --keep-up # leave the instance running afterwards
odysseus smoke --no-boot # drive whatever is already up here
odysseus smoke --restart # stop a running instance and boot fresh
odysseus smoke --areas # print the coverage table without running
odysseus smoke -- -k notes # everything after -- goes to pytest
The report is a per-area table, printed by the suite itself, listing the
areas it does not cover next to the ones it does. An area with no
scenario shows up as NOT RUN rather than going missing.
"""
from __future__ import annotations
import importlib.machinery
import importlib.util
import os
import subprocess
import sys
from pathlib import Path
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "_lib"))
from cli import quiet_logs, fail, common_parser, run # noqa: E402
quiet_logs()
SCRIPTS_DIR = Path(__file__).resolve().parent
REPO_ROOT = SCRIPTS_DIR.parent
# The launcher this tool delegates every instance decision to.
DEV_SCRIPT = "odysseus-dev"
# What pytest is pointed at, relative to the checkout root.
SMOKE_SUITE = "tests/smoke"
# Email is the one area with no reachable real backend, and the repo
# already has a deterministic path for it. Turning it on is the reason
# this tool owns the boot rather than leaving it to the caller: the flag
# is read inside the app's process, so it has to be in the environment
# the app is started with.
EMAIL_FIXTURE_ENV = "ODYSSEUS_EMAIL_FIXTURE"
def load_dev():
"""Import `scripts/odysseus-dev` as a module.
Same loader the CLI tests use. Delegating by import rather than by
parsing `odysseus dev env` output means the port derivation and the
credential handling have exactly one implementation.
"""
path = SCRIPTS_DIR / DEV_SCRIPT
if not path.exists():
fail(f"{path} is missing; this tool boots through it.", code=2)
loader = importlib.machinery.SourceFileLoader("odysseus_dev_cli", str(path))
spec = importlib.util.spec_from_loader(loader.name, loader)
module = importlib.util.module_from_spec(spec)
loader.exec_module(module)
return module
def suite_environment(dev, root, ports, account):
"""The environment the smoke suite reads its target instance from.
Deliberately the same values `odysseus dev env` prints, plus the dev
admin account, so a manual `pytest tests/smoke` under
`eval $(odysseus dev env)` behaves the way this tool does.
"""
data_dir = dev.dev_dir(root) / "data"
env = dict(os.environ)
env.update({
"APP_PORT": str(ports["app"]),
"CHROMADB_PORT": str(ports["chroma"]),
"ODYSSEUS_TEST_STATIC_PORT": str(ports["test_static"]),
"ODYSSEUS_DATA_DIR": str(data_dir),
"DATABASE_URL": f"sqlite:///{data_dir / 'app.db'}",
"ODYSSEUS_ADMIN_USER": account["username"],
"ODYSSEUS_ADMIN_PASSWORD": account["password"],
})
return env
def boot(dev, root, args):
"""Start the instance, or adopt one already running in this worktree.
Returns (started_by_us, note). A reused instance is never restarted
without being asked: it may be someone's debugging session, and the
one thing it can cost us is the email fixture flag, which the suite
reports as a skip rather than a pass.
"""
already = dev.running_app(dev.read_state(root))
if already and args.restart:
subprocess.run([sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "down"],
cwd=str(root), check=False)
already = None
if already:
return False, (
f"reusing the instance already up on port {already['port']} "
f"(pid {already['pid']}). If it was not booted with "
f"{EMAIL_FIXTURE_ENV}=1 the Email area will report a skip; "
f"re-run with --restart for a clean boot."
)
if args.no_boot:
fail(
"nothing is running in this worktree and --no-boot was passed.\n"
" boot it with `odysseus dev up`, or drop --no-boot.",
)
command = [sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "up"]
if args.venv:
command += ["--venv", args.venv]
env = dict(os.environ)
env[EMAIL_FIXTURE_ENV] = "1"
result = subprocess.run(command, cwd=str(root), env=env, check=False)
if result.returncode != 0:
fail(f"`odysseus dev up` exited {result.returncode}; not running the suite.")
return True, ""
def venv_python(dev, root, args):
"""The interpreter to run pytest with: the one the app runs under."""
recorded = (dev.read_state(root) or {}).get("venv")
for candidate in (Path(args.venv).expanduser() if args.venv else None,
Path(recorded) if recorded else None,
Path(root) / "venv"):
if candidate and (candidate / "bin" / "python").exists():
return candidate / "bin" / "python"
fail(
f"no interpreter found for the suite (looked at {Path(root) / 'venv'}).\n"
f" build one with ./start-macos.sh, or pass --venv."
)
def cmd_run(args):
dev = load_dev()
root = dev.find_repo_root(Path.cwd())
if root is None:
fail(f"not inside an Odysseus checkout (looked upwards from {Path.cwd()})", code=2)
if args.areas:
sys.path.insert(0, str(root))
from tests.smoke import areas
sys.stdout.write(areas.render_table({}, header="Odysseus release smoke - coverage") + "\n")
return 0
ports = dev.derive_ports(root)
account = dev.credentials(root)
started_by_us, note = boot(dev, root, args)
if note:
sys.stdout.write(f" {note}\n")
python = venv_python(dev, root, args)
env = suite_environment(dev, root, ports, account)
command = [str(python), "-m", "pytest", SMOKE_SUITE, "-q"] + list(args.pytest_args)
sys.stdout.write(f"\n running {SMOKE_SUITE} against http://127.0.0.1:{ports['app']}\n\n")
# Flush before handing the terminal to pytest, or our own lines land
# after its output and the report reads out of order.
sys.stdout.flush()
result = subprocess.run(command, cwd=str(root), env=env, check=False)
if started_by_us and not args.keep_up:
subprocess.run([sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "down"],
cwd=str(root), check=False)
elif started_by_us:
sys.stdout.write(
f"\n left running: http://127.0.0.1:{ports['app']} "
f"({account['username']} / {account['password']})\n"
f" stop it with `odysseus dev down`\n"
)
# `cli.run` discards a returned value but lets SystemExit through, and
# a smoke run's exit code is the whole point of having one command.
if result.returncode != 0:
raise SystemExit(result.returncode)
return 0
def build_parser():
parser = common_parser("odysseus-smoke",
"Boot this worktree and run the release smoke suite.")
parser.add_argument("--keep-up", action="store_true",
help="leave the instance running after the suite finishes")
parser.add_argument("--no-boot", action="store_true",
help="require an instance already up in this worktree")
parser.add_argument("--restart", action="store_true",
help="stop a running instance and boot a fresh one")
parser.add_argument("--venv", help="use this venv instead of ./venv")
parser.add_argument("--areas", action="store_true",
help="print the coverage table and exit without booting")
parser.add_argument("pytest_args", nargs="*", metavar="-- PYTEST ARGS",
help="arguments forwarded to pytest after a literal --")
parser.set_defaults(func=cmd_run)
return parser
if __name__ == "__main__":
sys.exit(run(build_parser()))
+25
View File
@@ -156,6 +156,31 @@ The inventory, the baseline and the capture live in `tests/css_snapshot/`;
not, and how to find the property that moved when it fails. The run takes about
21 seconds and skips when `npm ci` has not been run.
## Release smoke suite
`tests/smoke/` drives every advertised feature area once, end to end,
against a real instance - the safety net the unit suite does not provide
for a route move or a module split. One command boots the worktree and
runs it:
```bash
scripts/odysseus-smoke # boot, run every area, stop again
scripts/odysseus-smoke --keep-up # leave the instance running
scripts/odysseus-smoke --areas # the coverage table, without booting
```
It reads its target instance out of the environment (`APP_PORT` through
`internal_api_base()`, plus the dev admin account), so under a plain
`pytest` with nothing booted every scenario skips with the reason and
the full suite stays green. Models are served by a deterministic
loopback stub, never a live endpoint; email uses the repo's existing
`ODYSSEUS_EMAIL_FIXTURE` path.
The report is a per-area table that also prints the areas the suite
deliberately does not cover, so it cannot be read as coverage of
everything it omits. `tests/smoke/README.md` documents what is in each
list and why.
## Core principles
- Keep PRs small and homogeneous: one kind of change per PR.
+138
View File
@@ -0,0 +1,138 @@
# Release smoke suite
One command that boots this worktree and drives every advertised feature
area once, end to end, against a real instance.
```bash
scripts/odysseus-smoke # boot, run every area, stop again
scripts/odysseus-smoke --keep-up # leave the instance running afterwards
scripts/odysseus-smoke --no-boot # drive whatever is already up here
scripts/odysseus-smoke --areas # print the coverage table without booting
scripts/odysseus-smoke -- -k notes
```
## Why it exists
The decomposition work had two safety nets and neither covered the
product. The checkpoint benchmark measures the agent runtime. The
computed-style snapshot in `tests/test_css_computed_style_snapshot.py`
pins the rendered CSS. Nothing checked that Notes, Calendar, Documents,
Email, Memory, Cookbook or Settings still worked after a route package
moved or a 17,000-line module was split, and the unit suite does not:
`StressTestor`'s review of #5898 is the worked proof that a
byte-identical file-for-file move can break eleven tests that pass on
the base branch, with CI green throughout.
## Where it lives and why
pytest, not Playwright. Both are in the repo, so this adds no third
harness, and the choice went to pytest because every scenario here is a
request/response round trip rather than a rendering assertion -
rendering is already covered by the computed-style snapshot, and the
28 Playwright specs under `tests/e2e/photo-editor/` are the one area
with browser coverage. A browser would have added flake and start-up
cost for no extra signal.
It owns no instance logic. `scripts/odysseus-dev` already derives ports
per worktree, keeps the data dir and ChromaDB out of `data/`, and waits
on `/api/ready` rather than a TCP accept, so `scripts/odysseus-smoke`
boots through it and only adds the scenarios and the report.
## The contract with the runner
Four environment values, which are what `odysseus dev env` prints plus
the dev admin account:
| Variable | Read through | Used for |
|---|---|---|
| `APP_PORT` | `src.constants.internal_api_base()` | which instance to drive |
| `ODYSSEUS_ADMIN_USER` | - | who to authenticate as |
| `ODYSSEUS_ADMIN_PASSWORD` | - | " |
| `ODYSSEUS_DATA_DIR` | `src.constants.DATA_DIR` | where the email fixture file goes |
Run under a plain `pytest` with none of them set, every scenario skips
with the reason and the full suite stays green. `APP_PORT` pointing at
one of `odysseus dev`'s reserved ports - a normal launch of this
checkout, the machine's own instance - is refused rather than driven,
because the scenarios create and delete real records.
## The deterministic provider
`stub_provider.py` is an OpenAI-compatible server on an ephemeral
loopback port: `GET /v1/models` and `POST /v1/chat/completions`, both
buffered and streamed. No scenario touches a live model endpoint or the
network. It serves two model ids so the Compare area has something to
reveal, and it records every request so a scenario can assert the user's
message actually reached the provider rather than only that some text
came back.
Email uses the repo's own deterministic path rather than a second
mechanism: `routes/email_routes.py` serves a fixture inbox when
`ODYSSEUS_EMAIL_FIXTURE=1` and a fixture file is in the data dir. The
suite writes the file and restores whatever was there; the flag is read
inside the app's process, which is why the runner owns the boot.
## What is covered
One scenario per area, each asserting a user-visible outcome rather than
a status code. `scripts/odysseus-smoke --areas` prints the current list.
| Area | What it asserts |
|---|---|
| Chat | a turn against the stub comes back rendered, on both the buffered and the streamed path, and is in the session history |
| Compare | a blind comparison streams both sides and the vote reveals which model produced which reply |
| Notes | a note is listed, read back, edited, and 404s after delete |
| Calendar | an event appears in the window the UI queries and is gone after delete |
| Tasks | a daily task is accepted with a computed next run, is listed, and pauses |
| Documents (editor) | an edit adds a version, both versions read back, and a restore returns the first |
| Documents (RAG) | an uploaded file is chunked, indexed and listed |
| Email | the fixture inbox lists, opens with its body, and the unread count drops on mark-read |
| Memory | a fact is listed, found by search, and gone after delete |
| Uploads | an attachment reads back byte for byte |
| Cookbook | hardware is detected and recommendations come back sized against it; state persists |
| Settings | a preference written on one session is still there after a new login |
## What is not covered, and why
Printed next to the results on every run, so a reader cannot mistake the
table for coverage of everything it does not mention. `DECLARED_GAPS` in
`areas.py` is the list; the short version:
- **Deep Research** and **Web Search** need live egress. A deterministic
stub for the crawler would be an application change, which this is
not.
- **Email over IMAP/SMTP** is covered only as far as the fixture path
goes. There is no local mail server, so real account sync and send are
untested.
- **Cookbook download and serve** needs tmux, a GPU runtime and a
multi-gigabyte download.
- **Gallery and the photo editor** already have the repo's only
Playwright specs.
- **The agent tool loop** is what the checkpoint benchmark measures.
- **MCP servers** are stdio subprocesses outside the app's readiness
contract.
- **Rendering and layout** are pinned by the computed-style snapshot.
`Documents (RAG)` is the one covered area that can report `SKIP` on a
clean checkout: `requirements.txt` pins `chromadb-client`, the HTTP
client, and the ChromaDB *server* is a separate install. Without one
reachable, the upload route returns a deliberate 503 and the row reads
`SKIP` with that reason. Install `chromadb` in the venv and it goes
green.
## Reading the report
The table has one row per area in `areas.COVERED`, built from what
pytest reported rather than from anything a scenario asserts about
itself. An area whose module never ran shows as `NOT RUN`, so deleting
or renaming a file cannot make a row disappear -
`tests/test_smoke_area_table.py` pins that, and that a module on disk
must be registered.
## What a run leaves behind
Every scenario deletes what it created, with two exceptions on the
scratch instance: the preference key `odysseus_smoke_preference`, which
has no delete route, and the uploaded attachment, which the app's own
upload cleanup owns. Both live in `.odysseus-dev/data/`, never in
`data/`.
+195
View File
@@ -0,0 +1,195 @@
"""The area registry and the result table for the release smoke suite.
Pure stdlib on purpose: this module is the one part of the suite that has
to be readable and testable without a running instance, because it is
what decides whether the suite's output is honest.
Two lists matter here and they are both deliberate:
``COVERED`` names every feature area the suite drives, and the test
module that drives it. A row appears in the table whether or not its
module ran, so an area cannot quietly vanish from the report by having
its file deleted or renamed - it shows up as ``NOT RUN`` instead.
``DECLARED_GAPS`` names the areas the suite does *not* cover, with the
reason. They are printed alongside the results rather than left out,
because a smoke report that lists only what it checked reads as
coverage of everything it does not mention.
"""
from __future__ import annotations
import textwrap
from dataclasses import dataclass
# Result labels. ASCII only - no Unicode status glyphs anywhere in the
# table (repo convention: no emoji in UI or code).
PASS = "PASS"
FAIL = "FAIL"
SKIP = "SKIP"
NOT_RUN = "NOT RUN"
NOT_COVERED = "NOT COVERED"
# Precedence when one area's module produces several outcomes: a single
# failure decides the row, then a skip, then pass.
_PRECEDENCE = (FAIL, SKIP, PASS)
# Table geometry. Wide enough for the longest gap reason to read as a
# sentence, narrow enough to survive a normal terminal.
TABLE_WIDTH = 100
MIN_DETAIL_WIDTH = 30
@dataclass(frozen=True)
class Area:
"""One advertised feature area and the module that exercises it."""
key: str
label: str
module: str
@dataclass(frozen=True)
class Gap:
"""An area this suite does not cover, and why it does not."""
label: str
reason: str
# Order is the order the table prints in: the chat surface first, then
# the feature areas README.md advertises, then the setup surface.
COVERED = (
Area("chat", "Chat", "test_chat_smoke.py"),
Area("compare", "Compare", "test_compare_smoke.py"),
Area("notes", "Notes", "test_notes_smoke.py"),
Area("calendar", "Calendar", "test_calendar_smoke.py"),
Area("tasks", "Tasks (scheduled)", "test_tasks_smoke.py"),
Area("documents", "Documents (editor)", "test_documents_smoke.py"),
Area("documents_rag", "Documents (RAG)", "test_documents_rag_smoke.py"),
Area("email", "Email", "test_email_smoke.py"),
Area("memory", "Memory", "test_memory_smoke.py"),
Area("uploads", "Uploads", "test_uploads_smoke.py"),
Area("cookbook", "Cookbook", "test_cookbook_smoke.py"),
Area("settings", "Settings", "test_settings_smoke.py"),
)
DECLARED_GAPS = (
Gap(
"Deep Research",
"needs live web egress; the crawler has no deterministic stub and adding "
"one would be an application change",
),
Gap(
"Web Search",
"needs a reachable SearXNG or an external provider, so the result is not "
"reproducible from a clean checkout",
),
Gap(
"Email over IMAP/SMTP",
"covered through the existing ODYSSEUS_EMAIL_FIXTURE path only; no local "
"mail server, so real account sync and send are untested",
),
Gap(
"Cookbook download and serve",
"needs tmux, a GPU runtime and a multi-GB model download; only hardware "
"fit and state sync are checked",
),
Gap(
"Gallery and photo editor",
"already the one area with Playwright specs under tests/e2e/photo-editor/",
),
Gap(
"Agent tool loop",
"measured by the checkpoint benchmark, which is the safety net that does "
"cover the agent runtime",
),
Gap(
"MCP servers",
"the built-in servers are stdio subprocesses whose readiness is not part "
"of the app's own readiness contract",
),
Gap(
"Rendering and layout",
"pinned by the computed-style snapshot in "
"tests/test_css_computed_style_snapshot.py",
),
)
_MODULE_TO_KEY = {area.module: area.key for area in COVERED}
def area_for_module(module_name: str) -> str | None:
"""Map a test module filename to its area key, or None."""
return _MODULE_TO_KEY.get(module_name)
def resolve(outcomes: list[str]) -> str:
"""Collapse one module's outcomes into the row's single result."""
if not outcomes:
return NOT_RUN
for label in _PRECEDENCE:
if label in outcomes:
return label
return NOT_RUN
def render_table(results, *, header="", areas=COVERED, gaps=DECLARED_GAPS,
width=TABLE_WIDTH) -> str:
"""Render the per-area table.
``results`` maps an area key to a mapping with ``result`` and,
optionally, ``checks`` and ``detail``. Unknown keys are ignored and
missing keys render as ``NOT RUN`` - the registry, not the run,
decides which rows exist.
"""
rows = []
for area in areas:
entry = results.get(area.key) or {}
result = entry.get("result") or NOT_RUN
checks = entry.get("checks")
detail = entry.get("detail") or ""
if result == NOT_RUN and not detail:
detail = "no test ran for this area"
rows.append((area.label, result,
"" if checks is None else str(checks), detail))
labels = [row[0] for row in rows] + [gap.label for gap in gaps] + ["AREA"]
label_width = max(len(label) for label in labels)
result_width = max([len(row[1]) for row in rows] + [len(NOT_COVERED), len("RESULT")])
checks_width = max([len(row[2]) for row in rows] + [len("CHECKS")])
# Indent + label + gap + result + gap + checks + gap, then the detail.
detail_indent = 2 + label_width + 2 + result_width + 2 + checks_width + 2
detail_width = max(width - detail_indent, MIN_DETAIL_WIDTH)
def row_lines(label, result, checks, detail):
first = (f" {label.ljust(label_width)} {result.ljust(result_width)} "
f"{checks.rjust(checks_width)} ")
wrapped = textwrap.wrap(detail, detail_width) or [""]
out = [(first + wrapped[0]).rstrip()]
out += [(" " * detail_indent + line).rstrip() for line in wrapped[1:]]
return out
lines = []
if header:
lines.extend([header, ""])
lines.append(
f" {'AREA'.ljust(label_width)} {'RESULT'.ljust(result_width)} "
f"{'CHECKS'.rjust(checks_width)} DETAIL"
)
for row in rows:
lines.extend(row_lines(*row))
if gaps:
lines.extend(["", " Not covered, deliberately:"])
for gap in gaps:
lines.extend(row_lines(gap.label, NOT_COVERED, "", gap.reason))
failed = [row for row in rows if row[1] == FAIL]
skipped = [row for row in rows if row[1] == SKIP]
not_run = [row for row in rows if row[1] == NOT_RUN]
passed = len(rows) - len(failed) - len(skipped) - len(not_run)
lines.extend(["", (
f" {passed} pass, {len(failed)} fail, {len(skipped)} skip, "
f"{len(not_run)} not run, {len(gaps)} declared gaps"
)])
return "\n".join(lines)
+248
View File
@@ -0,0 +1,248 @@
"""Session wiring for the release smoke suite.
The suite drives a real instance over HTTP. It never starts one: that is
`scripts/odysseus-smoke`'s job, which boots the worktree through
`scripts/odysseus-dev` and hands the details over in the environment.
Run under a plain `pytest` with no instance up, every scenario skips
with the reason rather than failing, so the full suite stays green.
Three environment values form the contract, and they are exactly what
`odysseus dev env` prints plus the dev admin account:
APP_PORT - which instance, read through
`internal_api_base()`
ODYSSEUS_ADMIN_USER - the account to authenticate as
ODYSSEUS_ADMIN_PASSWORD
ODYSSEUS_DATA_DIR - where the email fixture file goes, read
through `src.constants.DATA_DIR`
"""
from __future__ import annotations
import os
import httpx
import pytest
from src.constants import internal_api_base
from tests.helpers.cli_loader import load_script
from tests.smoke import areas
from tests.smoke.stub_provider import MODEL_PRIMARY, StubProvider
# How long a smoke request may take. Generous: the first turn through a
# cold agent path does real work, and a timeout here reads as a product
# failure, which is the one thing this suite must not get wrong.
REQUEST_TIMEOUT_SECONDS = 120.0
# Auth and endpoint routes the suite drives directly. Kept here so a
# route rename shows up in one place rather than twelve.
LOGIN_PATH = "/api/auth/login"
HEALTH_PATH = "/api/health"
ENDPOINTS_PATH = "/api/model-endpoints"
SESSION_PATH = "/api/session"
_NO_PORT = (
"APP_PORT is not set, so there is no instance to drive. Run the suite "
"with `scripts/odysseus-smoke`, which boots this worktree and exports it."
)
def _reserved_ports() -> dict:
"""`odysseus dev`'s own refuse-list, read from the launcher.
The smoke suite writes and deletes real records, so pointing it at a
port that means something - a normal launch of this checkout, the
machine's production instance - has to be impossible rather than
merely discouraged. Reusing the launcher's table keeps one source of
truth instead of a second copy that can drift.
"""
try:
return dict(load_script("odysseus-dev").RESERVED_PORTS)
except Exception: # pragma: no cover - launcher absent or unloadable
return {}
@pytest.fixture(scope="session")
def base_url() -> str:
"""The instance this run drives, or a skip explaining why there is none."""
port = (os.environ.get("APP_PORT") or "").strip()
if not port:
pytest.skip(_NO_PORT)
reason = _reserved_ports().get(int(port)) if port.isdigit() else None
if reason:
pytest.skip(
f"APP_PORT={port} is {reason}. The smoke suite creates and deletes "
f"real records, so it refuses to run against that instance."
)
return internal_api_base()
@pytest.fixture(scope="session")
def account() -> dict:
user = (os.environ.get("ODYSSEUS_ADMIN_USER") or "").strip()
password = os.environ.get("ODYSSEUS_ADMIN_PASSWORD") or ""
if not user or not password:
pytest.skip(
"ODYSSEUS_ADMIN_USER / ODYSSEUS_ADMIN_PASSWORD are not set, so the "
"suite cannot authenticate. Run it with `scripts/odysseus-smoke`."
)
return {"username": user, "password": password}
def _new_client(base_url: str, account: dict) -> httpx.Client:
"""An authenticated client, or a skip naming what the instance said."""
client = httpx.Client(base_url=base_url, timeout=REQUEST_TIMEOUT_SECONDS,
follow_redirects=True)
try:
client.get(HEALTH_PATH)
except httpx.HTTPError as exc:
client.close()
pytest.skip(f"no instance answering at {base_url} ({exc}). Boot one with "
f"`odysseus dev up`, or run `scripts/odysseus-smoke`.")
response = client.post(LOGIN_PATH, json=account)
if response.status_code != 200:
client.close()
pytest.skip(
f"could not log in as {account['username']} at {base_url}: "
f"HTTP {response.status_code}. The recorded credentials may not "
f"match this instance's data dir."
)
return client
@pytest.fixture(scope="session")
def client(base_url, account):
"""One authenticated session shared by every scenario."""
handle = _new_client(base_url, account)
yield handle
handle.close()
@pytest.fixture
def fresh_client(base_url, account):
"""A second authenticated session, for asserting something persisted.
Reading a value back on the same cookie proves the request handler
returned it. Reading it back on a new login is the closest a test can
get to the user reloading the page.
"""
handle = _new_client(base_url, account)
yield handle
handle.close()
@pytest.fixture(scope="session")
def stub_provider():
"""The deterministic provider every model-backed scenario talks to."""
with StubProvider() as provider:
yield provider
@pytest.fixture(scope="session")
def stub_endpoint(client, stub_provider) -> str:
"""Register the stub as a model endpoint and return its id.
Registered as `endpoint_kind=local` so the app treats it the way it
treats a Cookbook-served model rather than probing it as a hosted
API, and removed afterwards so a `--keep-up` instance is not left
pointing at a port that has gone away.
"""
response = client.post(ENDPOINTS_PATH, data={
"name": "odysseus-smoke-stub",
"base_url": stub_provider.base_url,
"endpoint_kind": "local",
})
if response.status_code != 200:
pytest.skip(
f"the instance would not register the stub provider at "
f"{stub_provider.base_url}: HTTP {response.status_code} "
f"{response.text[:200]}"
)
body = response.json()
endpoint_id = str(body.get("id") or "")
if not endpoint_id:
pytest.skip(f"the endpoint the instance registered has no id: {body}")
if MODEL_PRIMARY not in (body.get("models") or []):
pytest.skip(
f"the instance did not discover {MODEL_PRIMARY} on the stub "
f"provider; it saw {body.get('models')}"
)
yield endpoint_id
client.delete(f"{ENDPOINTS_PATH}/{endpoint_id}")
@pytest.fixture
def chat_session(client, stub_endpoint):
"""A chat session bound to the stub provider, deleted afterwards."""
response = client.post(SESSION_PATH, data={
"name": "odysseus-smoke",
"endpoint_id": stub_endpoint,
"model": MODEL_PRIMARY,
})
assert response.status_code == 200, response.text
session_id = response.json()["id"]
yield session_id
client.delete(f"{SESSION_PATH}/{session_id}")
# --------------------------------------------------------------------------
# The per-area table
# --------------------------------------------------------------------------
# One row per area in `areas.COVERED`, built from the outcomes pytest
# reports rather than from anything a test asserts about itself, so a
# module that never ran cannot report a pass.
_outcomes: dict[str, list[str]] = {}
_details: dict[str, str] = {}
_checks: dict[str, int] = {}
def _skip_reason(report) -> str:
"""The reason text out of a skip report, best effort."""
longrepr = getattr(report, "longrepr", None)
if isinstance(longrepr, tuple) and len(longrepr) == 3:
reason = str(longrepr[2] or "")
return reason.removeprefix("Skipped: ").strip()
return str(longrepr or "").strip()
def pytest_runtest_logreport(report):
key = areas.area_for_module(os.path.basename(str(report.fspath)))
if key is None:
return
if report.skipped:
_outcomes.setdefault(key, []).append(areas.SKIP)
_details.setdefault(key, _skip_reason(report))
return
if report.failed:
_outcomes.setdefault(key, []).append(areas.FAIL)
_details[key] = f"{report.when} failed: {report.nodeid.split('::')[-1]}"
return
if report.when == "call" and report.passed:
_outcomes.setdefault(key, []).append(areas.PASS)
_checks[key] = _checks.get(key, 0) + 1
def pytest_terminal_summary(terminalreporter, exitstatus, config):
if not _outcomes:
return
results = {}
for key, outcomes in _outcomes.items():
results[key] = {
"result": areas.resolve(outcomes),
"checks": _checks.get(key, 0),
"detail": _details.get(key, ""),
}
if all(entry["result"] == areas.SKIP for entry in results.values()):
reasons = {entry["detail"] for entry in results.values() if entry["detail"]}
terminalreporter.write_line("")
terminalreporter.write_line(
"release smoke suite skipped: " + (
reasons.pop() if len(reasons) == 1 else "; ".join(sorted(reasons))
)
)
return
header = f"Odysseus release smoke - {internal_api_base()}"
terminalreporter.write_line("")
terminalreporter.write_line(areas.render_table(results, header=header))
+172
View File
@@ -0,0 +1,172 @@
"""A deterministic OpenAI-compatible provider for the smoke suite.
Every scenario that needs a model talks to this instead of a real
endpoint. It binds an ephemeral port on loopback, so no scenario depends
on network egress, on a model being downloaded, or on two runs on the
same machine picking the same port.
It answers the two routes the app needs to treat it as a local
OpenAI-compatible server: ``GET /v1/models`` for discovery and probing,
and ``POST /v1/chat/completions`` for both the buffered and the streamed
turn. Each reply is a fixed marker plus the model id, so a test can tell
the two models apart in a blind comparison; every request is recorded so
a test can assert the user's message actually reached the provider
rather than only that some text came back.
"""
from __future__ import annotations
import json
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
# Two model ids so the Compare area has something to reveal.
MODEL_PRIMARY = "odysseus-smoke-primary"
MODEL_SECONDARY = "odysseus-smoke-secondary"
MODELS = (MODEL_PRIMARY, MODEL_SECONDARY)
# The marker each reply starts with. Distinctive enough that finding it
# in a response body cannot be a coincidence, and short enough to read
# in a failure message.
REPLY_MARKER = "ODYSSEUS-SMOKE-REPLY"
# Bind on loopback, kernel-assigned port. No literal port anywhere.
BIND_HOST = "127.0.0.1"
BIND_PORT = 0
def reply_for(model: str) -> str:
"""The exact assistant text this provider returns for ``model``."""
return f"{REPLY_MARKER} {model}"
class _Recorder:
"""Requests the provider has served, for assertions after the fact."""
def __init__(self):
self._lock = threading.Lock()
self._calls = []
def record(self, payload: dict) -> None:
with self._lock:
self._calls.append(payload)
@property
def calls(self) -> list[dict]:
with self._lock:
return list(self._calls)
def prompts(self) -> list[str]:
"""Every user message this provider has been sent."""
out = []
for call in self.calls:
for message in call.get("messages") or []:
if message.get("role") == "user":
out.append(str(message.get("content") or ""))
return out
def clear(self) -> None:
with self._lock:
self._calls.clear()
def _handler_for(recorder: _Recorder):
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def log_message(self, *args): # noqa: D102 - silence stderr access log
pass
def _send_json(self, status: int, body: dict) -> None:
raw = json.dumps(body).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
self.wfile.write(raw)
def do_GET(self): # noqa: N802 - BaseHTTPRequestHandler's contract
if self.path.rstrip("/").endswith("/models"):
self._send_json(200, {
"object": "list",
"data": [{"id": name, "object": "model", "owned_by": "smoke"}
for name in MODELS],
})
return
self._send_json(404, {"error": {"message": f"no route {self.path}"}})
def do_POST(self): # noqa: N802 - BaseHTTPRequestHandler's contract
length = int(self.headers.get("Content-Length") or 0)
try:
payload = json.loads(self.rfile.read(length) or b"{}")
except ValueError:
payload = {}
if not isinstance(payload, dict):
payload = {}
recorder.record(payload)
model = str(payload.get("model") or MODEL_PRIMARY)
text = reply_for(model)
if payload.get("stream"):
self._send_stream(model, text)
return
self._send_json(200, {
"id": "smoke-completion",
"object": "chat.completion",
"model": model,
"choices": [{
"index": 0,
"message": {"role": "assistant", "content": text},
"finish_reason": "stop",
}],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
})
def _send_stream(self, model: str, text: str) -> None:
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Cache-Control", "no-cache")
self.send_header("Connection", "close")
self.end_headers()
for chunk in (
{"choices": [{"index": 0, "delta": {"content": text}}], "model": model},
{"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], "model": model},
):
self.wfile.write(b"data: " + json.dumps(chunk).encode("utf-8") + b"\n\n")
self.wfile.write(b"data: [DONE]\n\n")
self.wfile.flush()
return Handler
class StubProvider:
"""A running stub provider. Use as a context manager."""
def __init__(self):
self.recorder = _Recorder()
self._server = ThreadingHTTPServer((BIND_HOST, BIND_PORT), _handler_for(self.recorder))
self._server.daemon_threads = True
self._thread = threading.Thread(target=self._server.serve_forever, daemon=True)
@property
def port(self) -> int:
return self._server.server_address[1]
@property
def base_url(self) -> str:
"""The OpenAI-compatible base the app should be pointed at."""
return f"http://{BIND_HOST}:{self.port}/v1"
def start(self) -> "StubProvider":
self._thread.start()
return self
def stop(self) -> None:
self._server.shutdown()
self._server.server_close()
self._thread.join(timeout=5)
def __enter__(self) -> "StubProvider":
return self.start()
def __exit__(self, *exc) -> None:
self.stop()
+50
View File
@@ -0,0 +1,50 @@
"""Calendar: an event created through the API shows up in the range the UI asks for."""
from __future__ import annotations
from datetime import datetime, timedelta
CALENDARS_PATH = "/api/calendar/calendars"
EVENTS_PATH = "/api/calendar/events"
SUMMARY = "Odysseus smoke event"
# Far enough out that a real local calendar's own entries cannot collide
# with the assertion, and fixed relative to now so the window is never
# empty for date reasons.
DAYS_AHEAD = 30
def test_an_event_round_trips(client):
listed_calendars = client.get(CALENDARS_PATH)
assert listed_calendars.status_code == 200, listed_calendars.text
assert listed_calendars.json().get("calendars"), "no calendar to write an event into"
start = (datetime.now() + timedelta(days=DAYS_AHEAD)).replace(
hour=10, minute=0, second=0, microsecond=0)
created = client.post(EVENTS_PATH, json={
"summary": SUMMARY,
"dtstart": start.isoformat(),
})
assert created.status_code == 200, created.text
uid = created.json()["uid"]
try:
window = client.get(EVENTS_PATH, params={
"start": (start - timedelta(days=1)).isoformat(),
"end": (start + timedelta(days=1)).isoformat(),
})
assert window.status_code == 200, window.text
events = window.json().get("events") or []
matching = [e for e in events if e.get("uid") == uid]
assert matching, [e.get("summary") for e in events]
assert matching[0].get("summary") == SUMMARY, matching[0]
read = client.get(f"{EVENTS_PATH}/{uid}")
assert read.status_code == 200, read.text
finally:
removed = client.delete(f"{EVENTS_PATH}/{uid}")
assert removed.status_code == 200, removed.text
after = client.get(EVENTS_PATH, params={
"start": (start - timedelta(days=1)).isoformat(),
"end": (start + timedelta(days=1)).isoformat(),
})
assert uid not in [e.get("uid") for e in after.json().get("events") or []]
+66
View File
@@ -0,0 +1,66 @@
"""Chat: a turn against the stub provider comes back rendered and saved.
The buffered and the streamed path are both checked because the UI uses
the streamed one and the agent's own loop uses the buffered one, and a
decomposition can break either alone.
"""
from __future__ import annotations
import json
from tests.smoke.stub_provider import MODEL_PRIMARY, reply_for
CHAT_PATH = "/api/chat"
CHAT_STREAM_PATH = "/api/chat_stream"
HISTORY_PATH = "/api/history"
PROMPT = "Smoke check: reply with anything."
def test_buffered_turn_returns_the_provider_reply(client, chat_session, stub_provider):
response = client.post(CHAT_PATH, json={"message": PROMPT, "session": chat_session})
assert response.status_code == 200, response.text
body = response.json()
assert body.get("response") == reply_for(MODEL_PRIMARY), body
assert body.get("model") == MODEL_PRIMARY, body
# The app prefaces the turn with its own date/time context block, so
# the prompt is contained in what the provider saw rather than equal
# to it.
assert any(PROMPT in seen for seen in stub_provider.recorder.prompts()), (
"the prompt never reached the provider, so the reply came from "
"somewhere other than the model path"
)
def test_streamed_turn_emits_the_reply_and_saves_the_message(client, chat_session):
deltas, saved = [], []
with client.stream("POST", CHAT_STREAM_PATH,
json={"message": PROMPT, "session": chat_session}) as response:
assert response.status_code == 200
for line in response.iter_lines():
if not line.startswith("data: "):
continue
payload = line[len("data: "):].strip()
if payload == "[DONE]":
break
try:
event = json.loads(payload)
except ValueError:
continue
if "delta" in event:
deltas.append(str(event["delta"]))
if event.get("type") == "message_saved":
saved.append(event.get("id"))
assert "".join(deltas) == reply_for(MODEL_PRIMARY), deltas
assert saved and saved[0], "the stream never reported the assistant turn as saved"
def test_the_turn_is_in_the_session_history(client, chat_session):
client.post(CHAT_PATH, json={"message": PROMPT, "session": chat_session})
response = client.get(f"{HISTORY_PATH}/{chat_session}")
assert response.status_code == 200, response.text
messages = response.json().get("history") or []
rendered = [str(m.get("content") or "") for m in messages]
assert any(PROMPT in text for text in rendered), rendered
assert any(reply_for(MODEL_PRIMARY) in text for text in rendered), rendered
+71
View File
@@ -0,0 +1,71 @@
"""Compare: a blind comparison streams both sides and reveals them on the vote.
Two model ids on the one stub provider is what makes this checkable
without a second endpoint: each returns a reply naming itself, so the
reveal can be matched against which text arrived on which side.
"""
from __future__ import annotations
import json
from tests.smoke.stub_provider import MODEL_PRIMARY, MODEL_SECONDARY, reply_for
COMPARE_PATH = "/api/compare"
CHAT_STREAM_PATH = "/api/chat_stream"
PROMPT = "Smoke check: compare two replies."
def _stream_text(client, session_id: str) -> str:
deltas = []
with client.stream("POST", CHAT_STREAM_PATH,
json={"message": PROMPT, "session": session_id}) as response:
assert response.status_code == 200
for line in response.iter_lines():
if not line.startswith("data: "):
continue
payload = line[len("data: "):].strip()
if payload == "[DONE]":
break
try:
event = json.loads(payload)
except ValueError:
continue
if "delta" in event:
deltas.append(str(event["delta"]))
return "".join(deltas)
def test_a_blind_comparison_streams_and_reveals(client, stub_endpoint):
started = client.post(f"{COMPARE_PATH}/start", data={
"prompt": PROMPT,
"model_a": MODEL_PRIMARY,
"model_b": MODEL_SECONDARY,
"endpoint_a_id": stub_endpoint,
"endpoint_b_id": stub_endpoint,
"is_blind": "true",
})
assert started.status_code == 200, started.text
comparison = started.json()
comparison_id = comparison["id"]
# Blind: the start response must not say which model is on which side.
assert not comparison.get("model_left"), comparison
assert not comparison.get("model_right"), comparison
left = _stream_text(client, comparison["session_left"])
right = _stream_text(client, comparison["session_right"])
assert {left, right} == {reply_for(MODEL_PRIMARY), reply_for(MODEL_SECONDARY)}, (left, right)
voted = client.post(f"{COMPARE_PATH}/{comparison_id}/vote", data={"winner": "left"})
assert voted.status_code == 200, voted.text
revealed = voted.json().get("revealed") or {}
assert revealed.get("left") in (MODEL_PRIMARY, MODEL_SECONDARY), voted.text
assert reply_for(revealed["left"]) == left, (revealed, left)
assert reply_for(revealed["right"]) == right, (revealed, right)
history = client.get(f"{COMPARE_PATH}/history")
assert history.status_code == 200, history.text
entries = [row for row in history.json() if row.get("id") == comparison_id]
assert entries, history.text
assert entries[0].get("winner"), entries[0]
+50
View File
@@ -0,0 +1,50 @@
"""Cookbook: hardware is detected and the recommendations are sized against it.
What the README advertises here is hardware-aware recommendation, and
that is exactly the part that runs offline. Downloading and serving a
model is left to the gap list: it needs tmux, a GPU runtime and several
gigabytes over the network.
"""
from __future__ import annotations
SYSTEM_PATH = "/api/hwfit/system"
MODELS_PATH = "/api/hwfit/models"
STATE_PATH = "/api/cookbook/state"
GPUS_PATH = "/api/cookbook/gpus"
STATE_MARKER = "odysseusSmokeMarker"
def test_hardware_is_detected(client):
response = client.get(SYSTEM_PATH)
assert response.status_code == 200, response.text
system = response.json()
assert (system.get("total_ram_gb") or 0) > 0, system
assert (system.get("cpu_cores") or 0) > 0, system
assert system.get("cpu_name"), system
gpus = client.get(GPUS_PATH)
assert gpus.status_code == 200, gpus.text
assert gpus.json().get("ok") is True, gpus.text
def test_recommendations_fit_the_detected_hardware(client):
response = client.get(MODELS_PATH)
assert response.status_code == 200, response.text
body = response.json()
system = body.get("system") or {}
assert system.get("cpu_name"), body
recommended = body.get("models") or body.get("recommendations") or []
assert recommended, f"no model recommendation for this hardware: {list(body)}"
def test_cookbook_state_persists(client):
written = client.post(STATE_PATH, json={STATE_MARKER: "ody-95"})
assert written.status_code == 200, written.text
assert written.json().get("ok") is True, written.text
read = client.get(STATE_PATH)
assert read.status_code == 200, read.text
assert read.json().get(STATE_MARKER) == "ody-95", read.text
client.post(STATE_PATH, json={})
+56
View File
@@ -0,0 +1,56 @@
"""Documents (RAG): an uploaded file is chunked, indexed and then listed.
This is the one area whose dependency is not satisfiable from a clean
checkout. `requirements.txt` pins `chromadb-client`, the HTTP client;
the ChromaDB *server* is a separate install, and without one reachable
the app returns a deliberate 503 from the upload route rather than
indexing into nothing. So the scenario skips with that reason printed in
the table instead of being quietly dropped - a row saying SKIP and why
is the honest report, and it goes green as soon as a vector service is
there.
"""
from __future__ import annotations
import pytest
PERSONAL_PATH = "/api/personal"
UPLOAD_PATH = "/api/personal/upload"
# The route uniquifies the stored name and lists it under the owner's
# upload dir, so assertions match on the stem rather than the filename.
STEM = "odysseus-smoke-corpus"
FILENAME = f"{STEM}.txt"
CONTENT = (
"The release smoke suite indexed this file. "
"It exists so the retrieval path has something deterministic to chunk."
)
# The route's own 503 text when no vector store answers.
UNAVAILABLE_MARKER = "RAG system is not available"
def test_an_uploaded_file_is_indexed_and_listed(client):
response = client.post(UPLOAD_PATH,
files={"files": (FILENAME, CONTENT.encode("utf-8"), "text/plain")})
if response.status_code == 503 and UNAVAILABLE_MARKER in response.text:
pytest.skip(
"no vector service reachable, so indexing is unavailable. "
"requirements.txt pins chromadb-client, not the server; install "
"chromadb in the venv and re-run to cover this area."
)
assert response.status_code == 200, response.text
body = response.json()
try:
assert body.get("indexed_count", 0) > 0, f"nothing was indexed: {body}"
assert body.get("failed_count", 1) == 0, f"a chunk failed to index: {body}"
assert FILENAME in (body.get("uploaded") or []), body
listed = client.get(PERSONAL_PATH)
assert listed.status_code == 200, listed.text
names = [str(f.get("name")) for f in listed.json().get("files") or []]
assert any(STEM in name for name in names), names
finally:
listed = client.get(PERSONAL_PATH).json().get("files") or []
for entry in listed:
if STEM in str(entry.get("name")):
client.request("DELETE", "/api/personal/file",
params={"filepath": entry.get("path")})
+46
View File
@@ -0,0 +1,46 @@
"""Documents: the editor's create, edit and version history survive a round trip."""
from __future__ import annotations
DOCUMENT_PATH = "/api/document"
LIBRARY_PATH = "/api/documents/library"
TITLE = "Odysseus smoke document"
FIRST = "First revision, written by the release smoke suite."
SECOND = "Second revision, written by the release smoke suite."
def test_a_document_round_trips_with_its_versions(client):
created = client.post(DOCUMENT_PATH, json={"title": TITLE, "content": FIRST})
assert created.status_code == 200, created.text
body = created.json()
doc_id = body["id"]
try:
assert body.get("current_content") == FIRST, body
assert body.get("version_count") == 1, body
library = client.get(LIBRARY_PATH)
assert library.status_code == 200, library.text
assert doc_id in [d.get("id") for d in library.json().get("documents") or []]
# `force_version` because a save inside the route's coalesce
# window updates the current version in place instead of adding
# one - which is right for autosave and would make a smoke check
# that edits immediately depend on the clock.
edited = client.put(f"{DOCUMENT_PATH}/{doc_id}",
json={"content": SECOND, "force_version": True})
assert edited.status_code == 200, edited.text
assert edited.json().get("current_content") == SECOND, edited.text
assert edited.json().get("version_count") == 2, edited.text
versions = client.get(f"{DOCUMENT_PATH}/{doc_id}/versions")
assert versions.status_code == 200, versions.text
contents = {v.get("version_number"): v.get("content") for v in versions.json()}
assert contents.get(1) == FIRST, contents
assert contents.get(2) == SECOND, contents
restored = client.post(f"{DOCUMENT_PATH}/{doc_id}/restore/1")
assert restored.status_code == 200, restored.text
assert client.get(f"{DOCUMENT_PATH}/{doc_id}").json()["current_content"] == FIRST
finally:
removed = client.delete(f"{DOCUMENT_PATH}/{doc_id}")
assert removed.status_code == 200, removed.text
+96
View File
@@ -0,0 +1,96 @@
"""Email: the inbox lists a seeded message, opens it, and marks it read.
Email is the one area with no way to reach a real account deterministically,
and the repo already solved that: `routes/email_routes.py` carries a
fixture path gated on `ODYSSEUS_EMAIL_FIXTURE=1` plus a fixture file in
the data dir. This uses that mechanism rather than inventing a second
one - which means it also only covers what the fixture covers. Real IMAP
sync and SMTP send stay out, and say so in the table's gap list.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from src.constants import DATA_DIR
LIST_PATH = "/api/email/list"
READ_PATH = "/api/email/read"
MARK_READ_PATH = "/api/email/mark-read"
UNREAD_STATE_PATH = "/api/email/unread-state"
# The filename the fixture path reads. Same value as
# routes/email_routes.py's `_fixture_email_file`.
FIXTURE_FILENAME = "fixture_email_messages.json"
SUBJECT = "Odysseus smoke inbox message"
BODY = "Body of the smoke fixture message."
SENDER = "Smoke Sender <smoke@example.invalid>"
@pytest.fixture
def seeded_inbox(client, account):
"""Write the fixture inbox, and put back whatever was there before.
The flag itself has to be in the app's environment, which is the
launcher's job; if it is missing the fixture path stays off and the
list route falls through to a real account that does not exist. That
reads as a skip, not a failure.
"""
path = Path(DATA_DIR) / FIXTURE_FILENAME
previous = path.read_bytes() if path.exists() else None
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps({"messages": [{
"owner": account["username"],
"from": SENDER,
"subject": SUBJECT,
"date": "2026-09-29T12:00:00+00:00",
"body": BODY,
}]}, indent=2) + "\n", encoding="utf-8")
try:
yield path
finally:
if previous is None:
path.unlink(missing_ok=True)
else:
path.write_bytes(previous)
def _fixture_rows(client):
response = client.get(LIST_PATH, params={"folder": "INBOX", "limit": 10})
assert response.status_code == 200, response.text
body = response.json()
rows = [e for e in body.get("emails") or [] if e.get("subject") == SUBJECT]
if not rows:
pytest.skip(
"the instance is not serving the email fixture, so there is no "
"deterministic inbox to read. Boot it with ODYSSEUS_EMAIL_FIXTURE=1 "
"(scripts/odysseus-smoke does)."
)
return rows
def test_the_inbox_lists_opens_and_marks_a_message(client, seeded_inbox):
row = _fixture_rows(client)[0]
uid = row["uid"]
assert row.get("from_address") == "smoke@example.invalid", row
assert row.get("is_read") is False, row
read = client.get(f"{READ_PATH}/{uid}", params={"folder": "INBOX"})
assert read.status_code == 200, read.text
opened = read.json()
assert opened.get("subject") == SUBJECT, opened
assert BODY in str(opened.get("body") or ""), opened
assert BODY in str(opened.get("body_html") or ""), opened
before = client.get(UNREAD_STATE_PATH, params={"folder": "INBOX"})
assert before.status_code == 200, before.text
assert before.json().get("unread_count") == 1, before.text
marked = client.post(f"{MARK_READ_PATH}/{uid}", params={"folder": "INBOX"})
assert marked.status_code == 200, marked.text
after = client.get(UNREAD_STATE_PATH, params={"folder": "INBOX"})
assert after.json().get("unread_count") == 0, after.text
+40
View File
@@ -0,0 +1,40 @@
"""Memory: a stored fact is listed, found by search, and gone after delete.
Keyword mode is enough here on purpose. The memory store degrades to
keyword matching when no vector service answers, and that degraded path
is the one a clean checkout actually runs, so it is the one worth
smoking.
"""
from __future__ import annotations
MEMORY_PATH = "/api/memory"
ADD_PATH = "/api/memory/add"
SEARCH_PATH = "/api/memory/search"
# A token that cannot collide with a real memory on a scratch instance.
TOKEN = "odysseus-smoke-marker-quintile"
TEXT = f"The release smoke suite stored the token {TOKEN} as a fact."
def test_a_memory_round_trips(client):
created = client.post(ADD_PATH, json={"text": TEXT, "category": "fact"})
assert created.status_code == 200, created.text
assert created.json().get("ok") is True, created.text
listed = client.get(MEMORY_PATH)
assert listed.status_code == 200, listed.text
matching = [m for m in listed.json().get("memory") or [] if TOKEN in str(m.get("text"))]
assert matching, [m.get("text") for m in listed.json().get("memory") or []]
memory_id = matching[0]["id"]
try:
found = client.post(SEARCH_PATH, data={"query": TOKEN})
assert found.status_code == 200, found.text
hits = [m for m in found.json().get("memories") or [] if TOKEN in str(m.get("text"))]
assert hits, found.text
finally:
removed = client.delete(f"{MEMORY_PATH}/{memory_id}")
assert removed.status_code == 200, removed.text
remaining = client.get(MEMORY_PATH).json().get("memory") or []
assert memory_id not in [m.get("id") for m in remaining]
+32
View File
@@ -0,0 +1,32 @@
"""Notes: a note created through the API is readable, editable and gone after delete."""
from __future__ import annotations
NOTES_PATH = "/api/notes"
TITLE = "Odysseus smoke note"
BODY = "Created by the release smoke suite."
EDITED_BODY = "Edited by the release smoke suite."
def test_a_note_round_trips(client):
created = client.post(NOTES_PATH, json={"title": TITLE, "content": BODY})
assert created.status_code == 200, created.text
note_id = created.json()["id"]
try:
listed = client.get(NOTES_PATH)
assert listed.status_code == 200, listed.text
titles = [n.get("title") for n in listed.json().get("notes") or []]
assert TITLE in titles, titles
read = client.get(f"{NOTES_PATH}/{note_id}")
assert read.status_code == 200, read.text
assert read.json().get("content") == BODY, read.text
edited = client.put(f"{NOTES_PATH}/{note_id}",
json={"title": TITLE, "content": EDITED_BODY})
assert edited.status_code == 200, edited.text
assert client.get(f"{NOTES_PATH}/{note_id}").json()["content"] == EDITED_BODY
finally:
removed = client.delete(f"{NOTES_PATH}/{note_id}")
assert removed.status_code == 200, removed.text
assert client.get(f"{NOTES_PATH}/{note_id}").status_code == 404
+31
View File
@@ -0,0 +1,31 @@
"""Settings: a preference written through the API survives a new login.
Reading the value back on the same cookie only proves the handler
answered. Reading it back after authenticating again is what proves it
was persisted rather than held in the session, which is the closest an
API-level check gets to the user reloading the page.
"""
from __future__ import annotations
PREFS_PATH = "/api/prefs"
KEY = "odysseus_smoke_preference"
VALUE = "set-by-the-release-smoke-suite"
def test_a_preference_survives_a_new_login(client, fresh_client):
written = client.put(f"{PREFS_PATH}/{KEY}", json={"value": VALUE})
assert written.status_code == 200, written.text
assert written.json().get("value") == VALUE, written.text
read = client.get(f"{PREFS_PATH}/{KEY}")
assert read.status_code == 200, read.text
assert read.json().get("value") == VALUE, read.text
reloaded = fresh_client.get(f"{PREFS_PATH}/{KEY}")
assert reloaded.status_code == 200, reloaded.text
assert reloaded.json().get("value") == VALUE, reloaded.text
listed = fresh_client.get(PREFS_PATH)
assert listed.status_code == 200, listed.text
assert listed.json().get(KEY) == VALUE, listed.text
+41
View File
@@ -0,0 +1,41 @@
"""Tasks: a scheduled task is created with a computed next run and is listed.
Deliberately not fired. Running a task is model and tool work the
checkpoint benchmark covers; what this asserts is that the scheduler
still accepts a task and computes when it should run, which is the part
a route move can break silently.
"""
from __future__ import annotations
TASKS_PATH = "/api/tasks"
NAME = "Odysseus smoke task"
SCHEDULED_TIME = "03:00"
def test_a_scheduled_task_round_trips(client):
created = client.post(TASKS_PATH, json={
"name": NAME,
"task_type": "llm",
"prompt": "Smoke task; never run by this suite.",
"trigger_type": "schedule",
"schedule": "daily",
"scheduled_time": SCHEDULED_TIME,
})
assert created.status_code == 200, created.text
body = created.json()
task_id = body["id"]
try:
assert body.get("next_run"), f"no next run computed for a daily task: {body}"
assert body.get("status") == "active", body
listed = client.get(TASKS_PATH)
assert listed.status_code == 200, listed.text
assert task_id in [t.get("id") for t in listed.json().get("tasks") or []]
paused = client.post(f"{TASKS_PATH}/{task_id}/pause")
assert paused.status_code == 200, paused.text
assert client.get(f"{TASKS_PATH}/{task_id}").json().get("status") == "paused"
finally:
removed = client.delete(f"{TASKS_PATH}/{task_id}")
assert removed.status_code == 200, removed.text
+27
View File
@@ -0,0 +1,27 @@
"""Uploads: a file uploaded through the chat attachment route reads back byte for byte."""
from __future__ import annotations
UPLOAD_PATH = "/api/upload"
STATS_PATH = "/api/upload/stats"
FILENAME = "odysseus-smoke-attachment.txt"
CONTENT = b"Uploaded by the release smoke suite."
def test_an_upload_reads_back_unchanged(client):
response = client.post(UPLOAD_PATH,
files={"files": (FILENAME, CONTENT, "text/plain")})
assert response.status_code == 200, response.text
files = response.json().get("files") or []
assert len(files) == 1, response.text
entry = files[0]
assert entry.get("name") == FILENAME, entry
assert entry.get("size") == len(CONTENT), entry
fetched = client.get(f"{UPLOAD_PATH}/{entry['id']}")
assert fetched.status_code == 200, fetched.text
assert fetched.content == CONTENT, fetched.content
stats = client.get(STATS_PATH)
assert stats.status_code == 200, stats.text
assert stats.json().get("total_files", 0) >= 1, stats.text
+96
View File
@@ -0,0 +1,96 @@
"""The release smoke suite's report cannot overstate what it checked.
The suite's value is entirely in whether its table is honest, and the
table is built from a registry rather than from what happened to run.
These pin the properties that make it honest: every advertised area has
a row whether or not its module ran, every module that exists is
registered, a failure outranks a pass, and the whole thing stays ASCII.
"""
from pathlib import Path
import pytest
from tests.smoke import areas
SMOKE_DIR = Path(__file__).parent / "smoke"
def test_every_registered_area_has_its_module_on_disk():
missing = [area.module for area in areas.COVERED
if not (SMOKE_DIR / area.module).exists()]
assert not missing, f"registered areas with no test module: {missing}"
def test_every_smoke_module_is_registered():
"""A module nobody registered would run and never appear in the table."""
on_disk = {path.name for path in SMOKE_DIR.glob("test_*_smoke.py")}
registered = {area.module for area in areas.COVERED}
assert on_disk == registered, (
f"unregistered modules: {sorted(on_disk - registered)}; "
f"registered but absent: {sorted(registered - on_disk)}"
)
def test_area_keys_are_unique():
keys = [area.key for area in areas.COVERED]
assert len(keys) == len(set(keys)), keys
def test_a_run_that_reported_nothing_shows_every_area_as_not_run():
table = areas.render_table({})
for area in areas.COVERED:
assert area.label in table, area.label
assert table.count(areas.NOT_RUN) == len(areas.COVERED)
assert areas.PASS not in table
def test_an_unreported_area_is_not_dropped_from_the_table():
"""The registry decides the rows, so a partial run still lists the rest."""
table = areas.render_table({areas.COVERED[0].key: {"result": areas.PASS, "checks": 1}})
assert table.count(areas.NOT_RUN) == len(areas.COVERED) - 1
assert areas.COVERED[-1].label in table
def test_declared_gaps_are_printed_with_their_reason():
assert areas.DECLARED_GAPS, "a suite with no declared gaps is claiming total coverage"
table = areas.render_table({})
for gap in areas.DECLARED_GAPS:
assert gap.label in table, gap.label
# The reason is wrapped across lines, so match its first words.
assert " ".join(gap.reason.split()[:3]) in " ".join(table.split()), gap.reason
@pytest.mark.parametrize("outcomes,expected", [
([], areas.NOT_RUN),
([areas.PASS], areas.PASS),
([areas.PASS, areas.SKIP], areas.SKIP),
([areas.PASS, areas.SKIP, areas.FAIL], areas.FAIL),
([areas.PASS, areas.FAIL], areas.FAIL),
])
def test_the_worst_outcome_decides_the_row(outcomes, expected):
assert areas.resolve(outcomes) == expected
def test_the_table_is_ascii_only():
"""No emoji or status glyphs: the repo bans them in UI and in code."""
table = areas.render_table({area.key: {"result": areas.PASS, "checks": 1}
for area in areas.COVERED})
assert table.isascii(), [ch for ch in table if not ch.isascii()]
def test_module_names_map_back_to_their_area():
for area in areas.COVERED:
assert areas.area_for_module(area.module) == area.key
assert areas.area_for_module("test_not_a_smoke_module.py") is None
def test_the_summary_line_counts_every_row():
results = {area.key: {"result": areas.PASS, "checks": 1} for area in areas.COVERED}
results[areas.COVERED[0].key] = {"result": areas.FAIL, "checks": 0}
results[areas.COVERED[1].key] = {"result": areas.SKIP, "checks": 0}
summary = areas.render_table(results).splitlines()[-1]
assert f"{len(areas.COVERED) - 2} pass" in summary, summary
assert "1 fail" in summary, summary
assert "1 skip" in summary, summary
assert "0 not run" in summary, summary
assert f"{len(areas.DECLARED_GAPS)} declared gaps" in summary, summary