diff --git a/scripts/odysseus-smoke b/scripts/odysseus-smoke new file mode 100755 index 000000000..0534631d4 --- /dev/null +++ b/scripts/odysseus-smoke @@ -0,0 +1,209 @@ +#!/usr/bin/env python3 +"""odysseus-smoke — boot this worktree and drive every advertised feature area once. + +The decomposition work has two safety nets and neither one covers the +product: the checkpoint benchmark measures the agent runtime, and the +computed-style snapshot pins the CSS. Nothing checked that Notes, +Calendar, Documents, Email, Memory, Cookbook or Settings still worked +after a route package moved or a 17,000-line module was split. This is +that check, and it is deliberately shallow: one scenario per area, +asserting a user-visible outcome rather than an HTTP 200. + +It owns no instance logic. `odysseus dev` already isolates the ports, +the data dir and ChromaDB per worktree, so this boots through it, hands +the details to pytest in the environment, and stops what it started. + + odysseus smoke # boot, run every area, stop again + odysseus smoke --keep-up # leave the instance running afterwards + odysseus smoke --no-boot # drive whatever is already up here + odysseus smoke --restart # stop a running instance and boot fresh + odysseus smoke --areas # print the coverage table without running + odysseus smoke -- -k notes # everything after -- goes to pytest + +The report is a per-area table, printed by the suite itself, listing the +areas it does not cover next to the ones it does. An area with no +scenario shows up as NOT RUN rather than going missing. +""" +from __future__ import annotations + +import importlib.machinery +import importlib.util +import os +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "_lib")) +from cli import quiet_logs, fail, common_parser, run # noqa: E402 + +quiet_logs() + +SCRIPTS_DIR = Path(__file__).resolve().parent +REPO_ROOT = SCRIPTS_DIR.parent + +# The launcher this tool delegates every instance decision to. +DEV_SCRIPT = "odysseus-dev" + +# What pytest is pointed at, relative to the checkout root. +SMOKE_SUITE = "tests/smoke" + +# Email is the one area with no reachable real backend, and the repo +# already has a deterministic path for it. Turning it on is the reason +# this tool owns the boot rather than leaving it to the caller: the flag +# is read inside the app's process, so it has to be in the environment +# the app is started with. +EMAIL_FIXTURE_ENV = "ODYSSEUS_EMAIL_FIXTURE" + + +def load_dev(): + """Import `scripts/odysseus-dev` as a module. + + Same loader the CLI tests use. Delegating by import rather than by + parsing `odysseus dev env` output means the port derivation and the + credential handling have exactly one implementation. + """ + path = SCRIPTS_DIR / DEV_SCRIPT + if not path.exists(): + fail(f"{path} is missing; this tool boots through it.", code=2) + loader = importlib.machinery.SourceFileLoader("odysseus_dev_cli", str(path)) + spec = importlib.util.spec_from_loader(loader.name, loader) + module = importlib.util.module_from_spec(spec) + loader.exec_module(module) + return module + + +def suite_environment(dev, root, ports, account): + """The environment the smoke suite reads its target instance from. + + Deliberately the same values `odysseus dev env` prints, plus the dev + admin account, so a manual `pytest tests/smoke` under + `eval $(odysseus dev env)` behaves the way this tool does. + """ + data_dir = dev.dev_dir(root) / "data" + env = dict(os.environ) + env.update({ + "APP_PORT": str(ports["app"]), + "CHROMADB_PORT": str(ports["chroma"]), + "ODYSSEUS_TEST_STATIC_PORT": str(ports["test_static"]), + "ODYSSEUS_DATA_DIR": str(data_dir), + "DATABASE_URL": f"sqlite:///{data_dir / 'app.db'}", + "ODYSSEUS_ADMIN_USER": account["username"], + "ODYSSEUS_ADMIN_PASSWORD": account["password"], + }) + return env + + +def boot(dev, root, args): + """Start the instance, or adopt one already running in this worktree. + + Returns (started_by_us, note). A reused instance is never restarted + without being asked: it may be someone's debugging session, and the + one thing it can cost us is the email fixture flag, which the suite + reports as a skip rather than a pass. + """ + already = dev.running_app(dev.read_state(root)) + if already and args.restart: + subprocess.run([sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "down"], + cwd=str(root), check=False) + already = None + if already: + return False, ( + f"reusing the instance already up on port {already['port']} " + f"(pid {already['pid']}). If it was not booted with " + f"{EMAIL_FIXTURE_ENV}=1 the Email area will report a skip; " + f"re-run with --restart for a clean boot." + ) + if args.no_boot: + fail( + "nothing is running in this worktree and --no-boot was passed.\n" + " boot it with `odysseus dev up`, or drop --no-boot.", + ) + + command = [sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "up"] + if args.venv: + command += ["--venv", args.venv] + env = dict(os.environ) + env[EMAIL_FIXTURE_ENV] = "1" + result = subprocess.run(command, cwd=str(root), env=env, check=False) + if result.returncode != 0: + fail(f"`odysseus dev up` exited {result.returncode}; not running the suite.") + return True, "" + + +def venv_python(dev, root, args): + """The interpreter to run pytest with: the one the app runs under.""" + recorded = (dev.read_state(root) or {}).get("venv") + for candidate in (Path(args.venv).expanduser() if args.venv else None, + Path(recorded) if recorded else None, + Path(root) / "venv"): + if candidate and (candidate / "bin" / "python").exists(): + return candidate / "bin" / "python" + fail( + f"no interpreter found for the suite (looked at {Path(root) / 'venv'}).\n" + f" build one with ./start-macos.sh, or pass --venv." + ) + + +def cmd_run(args): + dev = load_dev() + root = dev.find_repo_root(Path.cwd()) + if root is None: + fail(f"not inside an Odysseus checkout (looked upwards from {Path.cwd()})", code=2) + + if args.areas: + sys.path.insert(0, str(root)) + from tests.smoke import areas + sys.stdout.write(areas.render_table({}, header="Odysseus release smoke - coverage") + "\n") + return 0 + + ports = dev.derive_ports(root) + account = dev.credentials(root) + started_by_us, note = boot(dev, root, args) + if note: + sys.stdout.write(f" {note}\n") + + python = venv_python(dev, root, args) + env = suite_environment(dev, root, ports, account) + command = [str(python), "-m", "pytest", SMOKE_SUITE, "-q"] + list(args.pytest_args) + sys.stdout.write(f"\n running {SMOKE_SUITE} against http://127.0.0.1:{ports['app']}\n\n") + # Flush before handing the terminal to pytest, or our own lines land + # after its output and the report reads out of order. + sys.stdout.flush() + result = subprocess.run(command, cwd=str(root), env=env, check=False) + + if started_by_us and not args.keep_up: + subprocess.run([sys.executable, str(SCRIPTS_DIR / DEV_SCRIPT), "down"], + cwd=str(root), check=False) + elif started_by_us: + sys.stdout.write( + f"\n left running: http://127.0.0.1:{ports['app']} " + f"({account['username']} / {account['password']})\n" + f" stop it with `odysseus dev down`\n" + ) + # `cli.run` discards a returned value but lets SystemExit through, and + # a smoke run's exit code is the whole point of having one command. + if result.returncode != 0: + raise SystemExit(result.returncode) + return 0 + + +def build_parser(): + parser = common_parser("odysseus-smoke", + "Boot this worktree and run the release smoke suite.") + parser.add_argument("--keep-up", action="store_true", + help="leave the instance running after the suite finishes") + parser.add_argument("--no-boot", action="store_true", + help="require an instance already up in this worktree") + parser.add_argument("--restart", action="store_true", + help="stop a running instance and boot a fresh one") + parser.add_argument("--venv", help="use this venv instead of ./venv") + parser.add_argument("--areas", action="store_true", + help="print the coverage table and exit without booting") + parser.add_argument("pytest_args", nargs="*", metavar="-- PYTEST ARGS", + help="arguments forwarded to pytest after a literal --") + parser.set_defaults(func=cmd_run) + return parser + + +if __name__ == "__main__": + sys.exit(run(build_parser())) diff --git a/tests/README.md b/tests/README.md index 085cb5f84..7acecaaad 100644 --- a/tests/README.md +++ b/tests/README.md @@ -156,6 +156,31 @@ The inventory, the baseline and the capture live in `tests/css_snapshot/`; not, and how to find the property that moved when it fails. The run takes about 21 seconds and skips when `npm ci` has not been run. +## Release smoke suite + +`tests/smoke/` drives every advertised feature area once, end to end, +against a real instance - the safety net the unit suite does not provide +for a route move or a module split. One command boots the worktree and +runs it: + +```bash +scripts/odysseus-smoke # boot, run every area, stop again +scripts/odysseus-smoke --keep-up # leave the instance running +scripts/odysseus-smoke --areas # the coverage table, without booting +``` + +It reads its target instance out of the environment (`APP_PORT` through +`internal_api_base()`, plus the dev admin account), so under a plain +`pytest` with nothing booted every scenario skips with the reason and +the full suite stays green. Models are served by a deterministic +loopback stub, never a live endpoint; email uses the repo's existing +`ODYSSEUS_EMAIL_FIXTURE` path. + +The report is a per-area table that also prints the areas the suite +deliberately does not cover, so it cannot be read as coverage of +everything it omits. `tests/smoke/README.md` documents what is in each +list and why. + ## Core principles - Keep PRs small and homogeneous: one kind of change per PR. diff --git a/tests/smoke/README.md b/tests/smoke/README.md new file mode 100644 index 000000000..dca572470 --- /dev/null +++ b/tests/smoke/README.md @@ -0,0 +1,138 @@ +# Release smoke suite + +One command that boots this worktree and drives every advertised feature +area once, end to end, against a real instance. + +```bash +scripts/odysseus-smoke # boot, run every area, stop again +scripts/odysseus-smoke --keep-up # leave the instance running afterwards +scripts/odysseus-smoke --no-boot # drive whatever is already up here +scripts/odysseus-smoke --areas # print the coverage table without booting +scripts/odysseus-smoke -- -k notes +``` + +## Why it exists + +The decomposition work had two safety nets and neither covered the +product. The checkpoint benchmark measures the agent runtime. The +computed-style snapshot in `tests/test_css_computed_style_snapshot.py` +pins the rendered CSS. Nothing checked that Notes, Calendar, Documents, +Email, Memory, Cookbook or Settings still worked after a route package +moved or a 17,000-line module was split, and the unit suite does not: +`StressTestor`'s review of #5898 is the worked proof that a +byte-identical file-for-file move can break eleven tests that pass on +the base branch, with CI green throughout. + +## Where it lives and why + +pytest, not Playwright. Both are in the repo, so this adds no third +harness, and the choice went to pytest because every scenario here is a +request/response round trip rather than a rendering assertion - +rendering is already covered by the computed-style snapshot, and the +28 Playwright specs under `tests/e2e/photo-editor/` are the one area +with browser coverage. A browser would have added flake and start-up +cost for no extra signal. + +It owns no instance logic. `scripts/odysseus-dev` already derives ports +per worktree, keeps the data dir and ChromaDB out of `data/`, and waits +on `/api/ready` rather than a TCP accept, so `scripts/odysseus-smoke` +boots through it and only adds the scenarios and the report. + +## The contract with the runner + +Four environment values, which are what `odysseus dev env` prints plus +the dev admin account: + +| Variable | Read through | Used for | +|---|---|---| +| `APP_PORT` | `src.constants.internal_api_base()` | which instance to drive | +| `ODYSSEUS_ADMIN_USER` | - | who to authenticate as | +| `ODYSSEUS_ADMIN_PASSWORD` | - | " | +| `ODYSSEUS_DATA_DIR` | `src.constants.DATA_DIR` | where the email fixture file goes | + +Run under a plain `pytest` with none of them set, every scenario skips +with the reason and the full suite stays green. `APP_PORT` pointing at +one of `odysseus dev`'s reserved ports - a normal launch of this +checkout, the machine's own instance - is refused rather than driven, +because the scenarios create and delete real records. + +## The deterministic provider + +`stub_provider.py` is an OpenAI-compatible server on an ephemeral +loopback port: `GET /v1/models` and `POST /v1/chat/completions`, both +buffered and streamed. No scenario touches a live model endpoint or the +network. It serves two model ids so the Compare area has something to +reveal, and it records every request so a scenario can assert the user's +message actually reached the provider rather than only that some text +came back. + +Email uses the repo's own deterministic path rather than a second +mechanism: `routes/email_routes.py` serves a fixture inbox when +`ODYSSEUS_EMAIL_FIXTURE=1` and a fixture file is in the data dir. The +suite writes the file and restores whatever was there; the flag is read +inside the app's process, which is why the runner owns the boot. + +## What is covered + +One scenario per area, each asserting a user-visible outcome rather than +a status code. `scripts/odysseus-smoke --areas` prints the current list. + +| Area | What it asserts | +|---|---| +| Chat | a turn against the stub comes back rendered, on both the buffered and the streamed path, and is in the session history | +| Compare | a blind comparison streams both sides and the vote reveals which model produced which reply | +| Notes | a note is listed, read back, edited, and 404s after delete | +| Calendar | an event appears in the window the UI queries and is gone after delete | +| Tasks | a daily task is accepted with a computed next run, is listed, and pauses | +| Documents (editor) | an edit adds a version, both versions read back, and a restore returns the first | +| Documents (RAG) | an uploaded file is chunked, indexed and listed | +| Email | the fixture inbox lists, opens with its body, and the unread count drops on mark-read | +| Memory | a fact is listed, found by search, and gone after delete | +| Uploads | an attachment reads back byte for byte | +| Cookbook | hardware is detected and recommendations come back sized against it; state persists | +| Settings | a preference written on one session is still there after a new login | + +## What is not covered, and why + +Printed next to the results on every run, so a reader cannot mistake the +table for coverage of everything it does not mention. `DECLARED_GAPS` in +`areas.py` is the list; the short version: + +- **Deep Research** and **Web Search** need live egress. A deterministic + stub for the crawler would be an application change, which this is + not. +- **Email over IMAP/SMTP** is covered only as far as the fixture path + goes. There is no local mail server, so real account sync and send are + untested. +- **Cookbook download and serve** needs tmux, a GPU runtime and a + multi-gigabyte download. +- **Gallery and the photo editor** already have the repo's only + Playwright specs. +- **The agent tool loop** is what the checkpoint benchmark measures. +- **MCP servers** are stdio subprocesses outside the app's readiness + contract. +- **Rendering and layout** are pinned by the computed-style snapshot. + +`Documents (RAG)` is the one covered area that can report `SKIP` on a +clean checkout: `requirements.txt` pins `chromadb-client`, the HTTP +client, and the ChromaDB *server* is a separate install. Without one +reachable, the upload route returns a deliberate 503 and the row reads +`SKIP` with that reason. Install `chromadb` in the venv and it goes +green. + +## Reading the report + +The table has one row per area in `areas.COVERED`, built from what +pytest reported rather than from anything a scenario asserts about +itself. An area whose module never ran shows as `NOT RUN`, so deleting +or renaming a file cannot make a row disappear - +`tests/test_smoke_area_table.py` pins that, and that a module on disk +must be registered. + +## What a run leaves behind + +Every scenario deletes what it created, with two exceptions on the +scratch instance: the preference key `odysseus_smoke_preference`, which +has no delete route, and the uploaded attachment, which the app's own +upload cleanup owns. Both live in `.odysseus-dev/data/`, never in +`data/`. diff --git a/tests/smoke/areas.py b/tests/smoke/areas.py new file mode 100644 index 000000000..8cdaf6296 --- /dev/null +++ b/tests/smoke/areas.py @@ -0,0 +1,195 @@ +"""The area registry and the result table for the release smoke suite. + +Pure stdlib on purpose: this module is the one part of the suite that has +to be readable and testable without a running instance, because it is +what decides whether the suite's output is honest. + +Two lists matter here and they are both deliberate: + +``COVERED`` names every feature area the suite drives, and the test +module that drives it. A row appears in the table whether or not its +module ran, so an area cannot quietly vanish from the report by having +its file deleted or renamed - it shows up as ``NOT RUN`` instead. + +``DECLARED_GAPS`` names the areas the suite does *not* cover, with the +reason. They are printed alongside the results rather than left out, +because a smoke report that lists only what it checked reads as +coverage of everything it does not mention. +""" +from __future__ import annotations + +import textwrap +from dataclasses import dataclass + +# Result labels. ASCII only - no Unicode status glyphs anywhere in the +# table (repo convention: no emoji in UI or code). +PASS = "PASS" +FAIL = "FAIL" +SKIP = "SKIP" +NOT_RUN = "NOT RUN" +NOT_COVERED = "NOT COVERED" + +# Precedence when one area's module produces several outcomes: a single +# failure decides the row, then a skip, then pass. +_PRECEDENCE = (FAIL, SKIP, PASS) + +# Table geometry. Wide enough for the longest gap reason to read as a +# sentence, narrow enough to survive a normal terminal. +TABLE_WIDTH = 100 +MIN_DETAIL_WIDTH = 30 + + +@dataclass(frozen=True) +class Area: + """One advertised feature area and the module that exercises it.""" + + key: str + label: str + module: str + + +@dataclass(frozen=True) +class Gap: + """An area this suite does not cover, and why it does not.""" + + label: str + reason: str + + +# Order is the order the table prints in: the chat surface first, then +# the feature areas README.md advertises, then the setup surface. +COVERED = ( + Area("chat", "Chat", "test_chat_smoke.py"), + Area("compare", "Compare", "test_compare_smoke.py"), + Area("notes", "Notes", "test_notes_smoke.py"), + Area("calendar", "Calendar", "test_calendar_smoke.py"), + Area("tasks", "Tasks (scheduled)", "test_tasks_smoke.py"), + Area("documents", "Documents (editor)", "test_documents_smoke.py"), + Area("documents_rag", "Documents (RAG)", "test_documents_rag_smoke.py"), + Area("email", "Email", "test_email_smoke.py"), + Area("memory", "Memory", "test_memory_smoke.py"), + Area("uploads", "Uploads", "test_uploads_smoke.py"), + Area("cookbook", "Cookbook", "test_cookbook_smoke.py"), + Area("settings", "Settings", "test_settings_smoke.py"), +) + +DECLARED_GAPS = ( + Gap( + "Deep Research", + "needs live web egress; the crawler has no deterministic stub and adding " + "one would be an application change", + ), + Gap( + "Web Search", + "needs a reachable SearXNG or an external provider, so the result is not " + "reproducible from a clean checkout", + ), + Gap( + "Email over IMAP/SMTP", + "covered through the existing ODYSSEUS_EMAIL_FIXTURE path only; no local " + "mail server, so real account sync and send are untested", + ), + Gap( + "Cookbook download and serve", + "needs tmux, a GPU runtime and a multi-GB model download; only hardware " + "fit and state sync are checked", + ), + Gap( + "Gallery and photo editor", + "already the one area with Playwright specs under tests/e2e/photo-editor/", + ), + Gap( + "Agent tool loop", + "measured by the checkpoint benchmark, which is the safety net that does " + "cover the agent runtime", + ), + Gap( + "MCP servers", + "the built-in servers are stdio subprocesses whose readiness is not part " + "of the app's own readiness contract", + ), + Gap( + "Rendering and layout", + "pinned by the computed-style snapshot in " + "tests/test_css_computed_style_snapshot.py", + ), +) + +_MODULE_TO_KEY = {area.module: area.key for area in COVERED} + + +def area_for_module(module_name: str) -> str | None: + """Map a test module filename to its area key, or None.""" + return _MODULE_TO_KEY.get(module_name) + + +def resolve(outcomes: list[str]) -> str: + """Collapse one module's outcomes into the row's single result.""" + if not outcomes: + return NOT_RUN + for label in _PRECEDENCE: + if label in outcomes: + return label + return NOT_RUN + + +def render_table(results, *, header="", areas=COVERED, gaps=DECLARED_GAPS, + width=TABLE_WIDTH) -> str: + """Render the per-area table. + + ``results`` maps an area key to a mapping with ``result`` and, + optionally, ``checks`` and ``detail``. Unknown keys are ignored and + missing keys render as ``NOT RUN`` - the registry, not the run, + decides which rows exist. + """ + rows = [] + for area in areas: + entry = results.get(area.key) or {} + result = entry.get("result") or NOT_RUN + checks = entry.get("checks") + detail = entry.get("detail") or "" + if result == NOT_RUN and not detail: + detail = "no test ran for this area" + rows.append((area.label, result, + "" if checks is None else str(checks), detail)) + + labels = [row[0] for row in rows] + [gap.label for gap in gaps] + ["AREA"] + label_width = max(len(label) for label in labels) + result_width = max([len(row[1]) for row in rows] + [len(NOT_COVERED), len("RESULT")]) + checks_width = max([len(row[2]) for row in rows] + [len("CHECKS")]) + # Indent + label + gap + result + gap + checks + gap, then the detail. + detail_indent = 2 + label_width + 2 + result_width + 2 + checks_width + 2 + detail_width = max(width - detail_indent, MIN_DETAIL_WIDTH) + + def row_lines(label, result, checks, detail): + first = (f" {label.ljust(label_width)} {result.ljust(result_width)} " + f"{checks.rjust(checks_width)} ") + wrapped = textwrap.wrap(detail, detail_width) or [""] + out = [(first + wrapped[0]).rstrip()] + out += [(" " * detail_indent + line).rstrip() for line in wrapped[1:]] + return out + + lines = [] + if header: + lines.extend([header, ""]) + lines.append( + f" {'AREA'.ljust(label_width)} {'RESULT'.ljust(result_width)} " + f"{'CHECKS'.rjust(checks_width)} DETAIL" + ) + for row in rows: + lines.extend(row_lines(*row)) + + if gaps: + lines.extend(["", " Not covered, deliberately:"]) + for gap in gaps: + lines.extend(row_lines(gap.label, NOT_COVERED, "", gap.reason)) + + failed = [row for row in rows if row[1] == FAIL] + skipped = [row for row in rows if row[1] == SKIP] + not_run = [row for row in rows if row[1] == NOT_RUN] + passed = len(rows) - len(failed) - len(skipped) - len(not_run) + lines.extend(["", ( + f" {passed} pass, {len(failed)} fail, {len(skipped)} skip, " + f"{len(not_run)} not run, {len(gaps)} declared gaps" + )]) + return "\n".join(lines) diff --git a/tests/smoke/conftest.py b/tests/smoke/conftest.py new file mode 100644 index 000000000..eb51fe58c --- /dev/null +++ b/tests/smoke/conftest.py @@ -0,0 +1,248 @@ +"""Session wiring for the release smoke suite. + +The suite drives a real instance over HTTP. It never starts one: that is +`scripts/odysseus-smoke`'s job, which boots the worktree through +`scripts/odysseus-dev` and hands the details over in the environment. +Run under a plain `pytest` with no instance up, every scenario skips +with the reason rather than failing, so the full suite stays green. + +Three environment values form the contract, and they are exactly what +`odysseus dev env` prints plus the dev admin account: + + APP_PORT - which instance, read through + `internal_api_base()` + ODYSSEUS_ADMIN_USER - the account to authenticate as + ODYSSEUS_ADMIN_PASSWORD + ODYSSEUS_DATA_DIR - where the email fixture file goes, read + through `src.constants.DATA_DIR` +""" +from __future__ import annotations + +import os + +import httpx +import pytest + +from src.constants import internal_api_base +from tests.helpers.cli_loader import load_script +from tests.smoke import areas +from tests.smoke.stub_provider import MODEL_PRIMARY, StubProvider + +# How long a smoke request may take. Generous: the first turn through a +# cold agent path does real work, and a timeout here reads as a product +# failure, which is the one thing this suite must not get wrong. +REQUEST_TIMEOUT_SECONDS = 120.0 + +# Auth and endpoint routes the suite drives directly. Kept here so a +# route rename shows up in one place rather than twelve. +LOGIN_PATH = "/api/auth/login" +HEALTH_PATH = "/api/health" +ENDPOINTS_PATH = "/api/model-endpoints" +SESSION_PATH = "/api/session" + +_NO_PORT = ( + "APP_PORT is not set, so there is no instance to drive. Run the suite " + "with `scripts/odysseus-smoke`, which boots this worktree and exports it." +) + + +def _reserved_ports() -> dict: + """`odysseus dev`'s own refuse-list, read from the launcher. + + The smoke suite writes and deletes real records, so pointing it at a + port that means something - a normal launch of this checkout, the + machine's production instance - has to be impossible rather than + merely discouraged. Reusing the launcher's table keeps one source of + truth instead of a second copy that can drift. + """ + try: + return dict(load_script("odysseus-dev").RESERVED_PORTS) + except Exception: # pragma: no cover - launcher absent or unloadable + return {} + + +@pytest.fixture(scope="session") +def base_url() -> str: + """The instance this run drives, or a skip explaining why there is none.""" + port = (os.environ.get("APP_PORT") or "").strip() + if not port: + pytest.skip(_NO_PORT) + reason = _reserved_ports().get(int(port)) if port.isdigit() else None + if reason: + pytest.skip( + f"APP_PORT={port} is {reason}. The smoke suite creates and deletes " + f"real records, so it refuses to run against that instance." + ) + return internal_api_base() + + +@pytest.fixture(scope="session") +def account() -> dict: + user = (os.environ.get("ODYSSEUS_ADMIN_USER") or "").strip() + password = os.environ.get("ODYSSEUS_ADMIN_PASSWORD") or "" + if not user or not password: + pytest.skip( + "ODYSSEUS_ADMIN_USER / ODYSSEUS_ADMIN_PASSWORD are not set, so the " + "suite cannot authenticate. Run it with `scripts/odysseus-smoke`." + ) + return {"username": user, "password": password} + + +def _new_client(base_url: str, account: dict) -> httpx.Client: + """An authenticated client, or a skip naming what the instance said.""" + client = httpx.Client(base_url=base_url, timeout=REQUEST_TIMEOUT_SECONDS, + follow_redirects=True) + try: + client.get(HEALTH_PATH) + except httpx.HTTPError as exc: + client.close() + pytest.skip(f"no instance answering at {base_url} ({exc}). Boot one with " + f"`odysseus dev up`, or run `scripts/odysseus-smoke`.") + response = client.post(LOGIN_PATH, json=account) + if response.status_code != 200: + client.close() + pytest.skip( + f"could not log in as {account['username']} at {base_url}: " + f"HTTP {response.status_code}. The recorded credentials may not " + f"match this instance's data dir." + ) + return client + + +@pytest.fixture(scope="session") +def client(base_url, account): + """One authenticated session shared by every scenario.""" + handle = _new_client(base_url, account) + yield handle + handle.close() + + +@pytest.fixture +def fresh_client(base_url, account): + """A second authenticated session, for asserting something persisted. + + Reading a value back on the same cookie proves the request handler + returned it. Reading it back on a new login is the closest a test can + get to the user reloading the page. + """ + handle = _new_client(base_url, account) + yield handle + handle.close() + + +@pytest.fixture(scope="session") +def stub_provider(): + """The deterministic provider every model-backed scenario talks to.""" + with StubProvider() as provider: + yield provider + + +@pytest.fixture(scope="session") +def stub_endpoint(client, stub_provider) -> str: + """Register the stub as a model endpoint and return its id. + + Registered as `endpoint_kind=local` so the app treats it the way it + treats a Cookbook-served model rather than probing it as a hosted + API, and removed afterwards so a `--keep-up` instance is not left + pointing at a port that has gone away. + """ + response = client.post(ENDPOINTS_PATH, data={ + "name": "odysseus-smoke-stub", + "base_url": stub_provider.base_url, + "endpoint_kind": "local", + }) + if response.status_code != 200: + pytest.skip( + f"the instance would not register the stub provider at " + f"{stub_provider.base_url}: HTTP {response.status_code} " + f"{response.text[:200]}" + ) + body = response.json() + endpoint_id = str(body.get("id") or "") + if not endpoint_id: + pytest.skip(f"the endpoint the instance registered has no id: {body}") + if MODEL_PRIMARY not in (body.get("models") or []): + pytest.skip( + f"the instance did not discover {MODEL_PRIMARY} on the stub " + f"provider; it saw {body.get('models')}" + ) + yield endpoint_id + client.delete(f"{ENDPOINTS_PATH}/{endpoint_id}") + + +@pytest.fixture +def chat_session(client, stub_endpoint): + """A chat session bound to the stub provider, deleted afterwards.""" + response = client.post(SESSION_PATH, data={ + "name": "odysseus-smoke", + "endpoint_id": stub_endpoint, + "model": MODEL_PRIMARY, + }) + assert response.status_code == 200, response.text + session_id = response.json()["id"] + yield session_id + client.delete(f"{SESSION_PATH}/{session_id}") + + +# -------------------------------------------------------------------------- +# The per-area table +# -------------------------------------------------------------------------- +# One row per area in `areas.COVERED`, built from the outcomes pytest +# reports rather than from anything a test asserts about itself, so a +# module that never ran cannot report a pass. + +_outcomes: dict[str, list[str]] = {} +_details: dict[str, str] = {} +_checks: dict[str, int] = {} + + +def _skip_reason(report) -> str: + """The reason text out of a skip report, best effort.""" + longrepr = getattr(report, "longrepr", None) + if isinstance(longrepr, tuple) and len(longrepr) == 3: + reason = str(longrepr[2] or "") + return reason.removeprefix("Skipped: ").strip() + return str(longrepr or "").strip() + + +def pytest_runtest_logreport(report): + key = areas.area_for_module(os.path.basename(str(report.fspath))) + if key is None: + return + if report.skipped: + _outcomes.setdefault(key, []).append(areas.SKIP) + _details.setdefault(key, _skip_reason(report)) + return + if report.failed: + _outcomes.setdefault(key, []).append(areas.FAIL) + _details[key] = f"{report.when} failed: {report.nodeid.split('::')[-1]}" + return + if report.when == "call" and report.passed: + _outcomes.setdefault(key, []).append(areas.PASS) + _checks[key] = _checks.get(key, 0) + 1 + + +def pytest_terminal_summary(terminalreporter, exitstatus, config): + if not _outcomes: + return + results = {} + for key, outcomes in _outcomes.items(): + results[key] = { + "result": areas.resolve(outcomes), + "checks": _checks.get(key, 0), + "detail": _details.get(key, ""), + } + + if all(entry["result"] == areas.SKIP for entry in results.values()): + reasons = {entry["detail"] for entry in results.values() if entry["detail"]} + terminalreporter.write_line("") + terminalreporter.write_line( + "release smoke suite skipped: " + ( + reasons.pop() if len(reasons) == 1 else "; ".join(sorted(reasons)) + ) + ) + return + + header = f"Odysseus release smoke - {internal_api_base()}" + terminalreporter.write_line("") + terminalreporter.write_line(areas.render_table(results, header=header)) diff --git a/tests/smoke/stub_provider.py b/tests/smoke/stub_provider.py new file mode 100644 index 000000000..ef007f89c --- /dev/null +++ b/tests/smoke/stub_provider.py @@ -0,0 +1,172 @@ +"""A deterministic OpenAI-compatible provider for the smoke suite. + +Every scenario that needs a model talks to this instead of a real +endpoint. It binds an ephemeral port on loopback, so no scenario depends +on network egress, on a model being downloaded, or on two runs on the +same machine picking the same port. + +It answers the two routes the app needs to treat it as a local +OpenAI-compatible server: ``GET /v1/models`` for discovery and probing, +and ``POST /v1/chat/completions`` for both the buffered and the streamed +turn. Each reply is a fixed marker plus the model id, so a test can tell +the two models apart in a blind comparison; every request is recorded so +a test can assert the user's message actually reached the provider +rather than only that some text came back. +""" +from __future__ import annotations + +import json +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +# Two model ids so the Compare area has something to reveal. +MODEL_PRIMARY = "odysseus-smoke-primary" +MODEL_SECONDARY = "odysseus-smoke-secondary" +MODELS = (MODEL_PRIMARY, MODEL_SECONDARY) + +# The marker each reply starts with. Distinctive enough that finding it +# in a response body cannot be a coincidence, and short enough to read +# in a failure message. +REPLY_MARKER = "ODYSSEUS-SMOKE-REPLY" + +# Bind on loopback, kernel-assigned port. No literal port anywhere. +BIND_HOST = "127.0.0.1" +BIND_PORT = 0 + + +def reply_for(model: str) -> str: + """The exact assistant text this provider returns for ``model``.""" + return f"{REPLY_MARKER} {model}" + + +class _Recorder: + """Requests the provider has served, for assertions after the fact.""" + + def __init__(self): + self._lock = threading.Lock() + self._calls = [] + + def record(self, payload: dict) -> None: + with self._lock: + self._calls.append(payload) + + @property + def calls(self) -> list[dict]: + with self._lock: + return list(self._calls) + + def prompts(self) -> list[str]: + """Every user message this provider has been sent.""" + out = [] + for call in self.calls: + for message in call.get("messages") or []: + if message.get("role") == "user": + out.append(str(message.get("content") or "")) + return out + + def clear(self) -> None: + with self._lock: + self._calls.clear() + + +def _handler_for(recorder: _Recorder): + class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, *args): # noqa: D102 - silence stderr access log + pass + + def _send_json(self, status: int, body: dict) -> None: + raw = json.dumps(body).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + def do_GET(self): # noqa: N802 - BaseHTTPRequestHandler's contract + if self.path.rstrip("/").endswith("/models"): + self._send_json(200, { + "object": "list", + "data": [{"id": name, "object": "model", "owned_by": "smoke"} + for name in MODELS], + }) + return + self._send_json(404, {"error": {"message": f"no route {self.path}"}}) + + def do_POST(self): # noqa: N802 - BaseHTTPRequestHandler's contract + length = int(self.headers.get("Content-Length") or 0) + try: + payload = json.loads(self.rfile.read(length) or b"{}") + except ValueError: + payload = {} + if not isinstance(payload, dict): + payload = {} + recorder.record(payload) + + model = str(payload.get("model") or MODEL_PRIMARY) + text = reply_for(model) + if payload.get("stream"): + self._send_stream(model, text) + return + self._send_json(200, { + "id": "smoke-completion", + "object": "chat.completion", + "model": model, + "choices": [{ + "index": 0, + "message": {"role": "assistant", "content": text}, + "finish_reason": "stop", + }], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }) + + def _send_stream(self, model: str, text: str) -> None: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-cache") + self.send_header("Connection", "close") + self.end_headers() + for chunk in ( + {"choices": [{"index": 0, "delta": {"content": text}}], "model": model}, + {"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], "model": model}, + ): + self.wfile.write(b"data: " + json.dumps(chunk).encode("utf-8") + b"\n\n") + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + + return Handler + + +class StubProvider: + """A running stub provider. Use as a context manager.""" + + def __init__(self): + self.recorder = _Recorder() + self._server = ThreadingHTTPServer((BIND_HOST, BIND_PORT), _handler_for(self.recorder)) + self._server.daemon_threads = True + self._thread = threading.Thread(target=self._server.serve_forever, daemon=True) + + @property + def port(self) -> int: + return self._server.server_address[1] + + @property + def base_url(self) -> str: + """The OpenAI-compatible base the app should be pointed at.""" + return f"http://{BIND_HOST}:{self.port}/v1" + + def start(self) -> "StubProvider": + self._thread.start() + return self + + def stop(self) -> None: + self._server.shutdown() + self._server.server_close() + self._thread.join(timeout=5) + + def __enter__(self) -> "StubProvider": + return self.start() + + def __exit__(self, *exc) -> None: + self.stop() diff --git a/tests/smoke/test_calendar_smoke.py b/tests/smoke/test_calendar_smoke.py new file mode 100644 index 000000000..34567d62f --- /dev/null +++ b/tests/smoke/test_calendar_smoke.py @@ -0,0 +1,50 @@ +"""Calendar: an event created through the API shows up in the range the UI asks for.""" +from __future__ import annotations + +from datetime import datetime, timedelta + +CALENDARS_PATH = "/api/calendar/calendars" +EVENTS_PATH = "/api/calendar/events" + +SUMMARY = "Odysseus smoke event" +# Far enough out that a real local calendar's own entries cannot collide +# with the assertion, and fixed relative to now so the window is never +# empty for date reasons. +DAYS_AHEAD = 30 + + +def test_an_event_round_trips(client): + listed_calendars = client.get(CALENDARS_PATH) + assert listed_calendars.status_code == 200, listed_calendars.text + assert listed_calendars.json().get("calendars"), "no calendar to write an event into" + + start = (datetime.now() + timedelta(days=DAYS_AHEAD)).replace( + hour=10, minute=0, second=0, microsecond=0) + created = client.post(EVENTS_PATH, json={ + "summary": SUMMARY, + "dtstart": start.isoformat(), + }) + assert created.status_code == 200, created.text + uid = created.json()["uid"] + try: + window = client.get(EVENTS_PATH, params={ + "start": (start - timedelta(days=1)).isoformat(), + "end": (start + timedelta(days=1)).isoformat(), + }) + assert window.status_code == 200, window.text + events = window.json().get("events") or [] + matching = [e for e in events if e.get("uid") == uid] + assert matching, [e.get("summary") for e in events] + assert matching[0].get("summary") == SUMMARY, matching[0] + + read = client.get(f"{EVENTS_PATH}/{uid}") + assert read.status_code == 200, read.text + finally: + removed = client.delete(f"{EVENTS_PATH}/{uid}") + assert removed.status_code == 200, removed.text + + after = client.get(EVENTS_PATH, params={ + "start": (start - timedelta(days=1)).isoformat(), + "end": (start + timedelta(days=1)).isoformat(), + }) + assert uid not in [e.get("uid") for e in after.json().get("events") or []] diff --git a/tests/smoke/test_chat_smoke.py b/tests/smoke/test_chat_smoke.py new file mode 100644 index 000000000..32a266871 --- /dev/null +++ b/tests/smoke/test_chat_smoke.py @@ -0,0 +1,66 @@ +"""Chat: a turn against the stub provider comes back rendered and saved. + +The buffered and the streamed path are both checked because the UI uses +the streamed one and the agent's own loop uses the buffered one, and a +decomposition can break either alone. +""" +from __future__ import annotations + +import json + +from tests.smoke.stub_provider import MODEL_PRIMARY, reply_for + +CHAT_PATH = "/api/chat" +CHAT_STREAM_PATH = "/api/chat_stream" +HISTORY_PATH = "/api/history" + +PROMPT = "Smoke check: reply with anything." + + +def test_buffered_turn_returns_the_provider_reply(client, chat_session, stub_provider): + response = client.post(CHAT_PATH, json={"message": PROMPT, "session": chat_session}) + assert response.status_code == 200, response.text + body = response.json() + assert body.get("response") == reply_for(MODEL_PRIMARY), body + assert body.get("model") == MODEL_PRIMARY, body + # The app prefaces the turn with its own date/time context block, so + # the prompt is contained in what the provider saw rather than equal + # to it. + assert any(PROMPT in seen for seen in stub_provider.recorder.prompts()), ( + "the prompt never reached the provider, so the reply came from " + "somewhere other than the model path" + ) + + +def test_streamed_turn_emits_the_reply_and_saves_the_message(client, chat_session): + deltas, saved = [], [] + with client.stream("POST", CHAT_STREAM_PATH, + json={"message": PROMPT, "session": chat_session}) as response: + assert response.status_code == 200 + for line in response.iter_lines(): + if not line.startswith("data: "): + continue + payload = line[len("data: "):].strip() + if payload == "[DONE]": + break + try: + event = json.loads(payload) + except ValueError: + continue + if "delta" in event: + deltas.append(str(event["delta"])) + if event.get("type") == "message_saved": + saved.append(event.get("id")) + + assert "".join(deltas) == reply_for(MODEL_PRIMARY), deltas + assert saved and saved[0], "the stream never reported the assistant turn as saved" + + +def test_the_turn_is_in_the_session_history(client, chat_session): + client.post(CHAT_PATH, json={"message": PROMPT, "session": chat_session}) + response = client.get(f"{HISTORY_PATH}/{chat_session}") + assert response.status_code == 200, response.text + messages = response.json().get("history") or [] + rendered = [str(m.get("content") or "") for m in messages] + assert any(PROMPT in text for text in rendered), rendered + assert any(reply_for(MODEL_PRIMARY) in text for text in rendered), rendered diff --git a/tests/smoke/test_compare_smoke.py b/tests/smoke/test_compare_smoke.py new file mode 100644 index 000000000..af7ef18ff --- /dev/null +++ b/tests/smoke/test_compare_smoke.py @@ -0,0 +1,71 @@ +"""Compare: a blind comparison streams both sides and reveals them on the vote. + +Two model ids on the one stub provider is what makes this checkable +without a second endpoint: each returns a reply naming itself, so the +reveal can be matched against which text arrived on which side. +""" +from __future__ import annotations + +import json + +from tests.smoke.stub_provider import MODEL_PRIMARY, MODEL_SECONDARY, reply_for + +COMPARE_PATH = "/api/compare" +CHAT_STREAM_PATH = "/api/chat_stream" + +PROMPT = "Smoke check: compare two replies." + + +def _stream_text(client, session_id: str) -> str: + deltas = [] + with client.stream("POST", CHAT_STREAM_PATH, + json={"message": PROMPT, "session": session_id}) as response: + assert response.status_code == 200 + for line in response.iter_lines(): + if not line.startswith("data: "): + continue + payload = line[len("data: "):].strip() + if payload == "[DONE]": + break + try: + event = json.loads(payload) + except ValueError: + continue + if "delta" in event: + deltas.append(str(event["delta"])) + return "".join(deltas) + + +def test_a_blind_comparison_streams_and_reveals(client, stub_endpoint): + started = client.post(f"{COMPARE_PATH}/start", data={ + "prompt": PROMPT, + "model_a": MODEL_PRIMARY, + "model_b": MODEL_SECONDARY, + "endpoint_a_id": stub_endpoint, + "endpoint_b_id": stub_endpoint, + "is_blind": "true", + }) + assert started.status_code == 200, started.text + comparison = started.json() + comparison_id = comparison["id"] + + # Blind: the start response must not say which model is on which side. + assert not comparison.get("model_left"), comparison + assert not comparison.get("model_right"), comparison + + left = _stream_text(client, comparison["session_left"]) + right = _stream_text(client, comparison["session_right"]) + assert {left, right} == {reply_for(MODEL_PRIMARY), reply_for(MODEL_SECONDARY)}, (left, right) + + voted = client.post(f"{COMPARE_PATH}/{comparison_id}/vote", data={"winner": "left"}) + assert voted.status_code == 200, voted.text + revealed = voted.json().get("revealed") or {} + assert revealed.get("left") in (MODEL_PRIMARY, MODEL_SECONDARY), voted.text + assert reply_for(revealed["left"]) == left, (revealed, left) + assert reply_for(revealed["right"]) == right, (revealed, right) + + history = client.get(f"{COMPARE_PATH}/history") + assert history.status_code == 200, history.text + entries = [row for row in history.json() if row.get("id") == comparison_id] + assert entries, history.text + assert entries[0].get("winner"), entries[0] diff --git a/tests/smoke/test_cookbook_smoke.py b/tests/smoke/test_cookbook_smoke.py new file mode 100644 index 000000000..5ee679b58 --- /dev/null +++ b/tests/smoke/test_cookbook_smoke.py @@ -0,0 +1,50 @@ +"""Cookbook: hardware is detected and the recommendations are sized against it. + +What the README advertises here is hardware-aware recommendation, and +that is exactly the part that runs offline. Downloading and serving a +model is left to the gap list: it needs tmux, a GPU runtime and several +gigabytes over the network. +""" +from __future__ import annotations + +SYSTEM_PATH = "/api/hwfit/system" +MODELS_PATH = "/api/hwfit/models" +STATE_PATH = "/api/cookbook/state" +GPUS_PATH = "/api/cookbook/gpus" + +STATE_MARKER = "odysseusSmokeMarker" + + +def test_hardware_is_detected(client): + response = client.get(SYSTEM_PATH) + assert response.status_code == 200, response.text + system = response.json() + assert (system.get("total_ram_gb") or 0) > 0, system + assert (system.get("cpu_cores") or 0) > 0, system + assert system.get("cpu_name"), system + + gpus = client.get(GPUS_PATH) + assert gpus.status_code == 200, gpus.text + assert gpus.json().get("ok") is True, gpus.text + + +def test_recommendations_fit_the_detected_hardware(client): + response = client.get(MODELS_PATH) + assert response.status_code == 200, response.text + body = response.json() + system = body.get("system") or {} + assert system.get("cpu_name"), body + recommended = body.get("models") or body.get("recommendations") or [] + assert recommended, f"no model recommendation for this hardware: {list(body)}" + + +def test_cookbook_state_persists(client): + written = client.post(STATE_PATH, json={STATE_MARKER: "ody-95"}) + assert written.status_code == 200, written.text + assert written.json().get("ok") is True, written.text + + read = client.get(STATE_PATH) + assert read.status_code == 200, read.text + assert read.json().get(STATE_MARKER) == "ody-95", read.text + + client.post(STATE_PATH, json={}) diff --git a/tests/smoke/test_documents_rag_smoke.py b/tests/smoke/test_documents_rag_smoke.py new file mode 100644 index 000000000..ee2bd1e89 --- /dev/null +++ b/tests/smoke/test_documents_rag_smoke.py @@ -0,0 +1,56 @@ +"""Documents (RAG): an uploaded file is chunked, indexed and then listed. + +This is the one area whose dependency is not satisfiable from a clean +checkout. `requirements.txt` pins `chromadb-client`, the HTTP client; +the ChromaDB *server* is a separate install, and without one reachable +the app returns a deliberate 503 from the upload route rather than +indexing into nothing. So the scenario skips with that reason printed in +the table instead of being quietly dropped - a row saying SKIP and why +is the honest report, and it goes green as soon as a vector service is +there. +""" +from __future__ import annotations + +import pytest + +PERSONAL_PATH = "/api/personal" +UPLOAD_PATH = "/api/personal/upload" + +# The route uniquifies the stored name and lists it under the owner's +# upload dir, so assertions match on the stem rather than the filename. +STEM = "odysseus-smoke-corpus" +FILENAME = f"{STEM}.txt" +CONTENT = ( + "The release smoke suite indexed this file. " + "It exists so the retrieval path has something deterministic to chunk." +) +# The route's own 503 text when no vector store answers. +UNAVAILABLE_MARKER = "RAG system is not available" + + +def test_an_uploaded_file_is_indexed_and_listed(client): + response = client.post(UPLOAD_PATH, + files={"files": (FILENAME, CONTENT.encode("utf-8"), "text/plain")}) + if response.status_code == 503 and UNAVAILABLE_MARKER in response.text: + pytest.skip( + "no vector service reachable, so indexing is unavailable. " + "requirements.txt pins chromadb-client, not the server; install " + "chromadb in the venv and re-run to cover this area." + ) + assert response.status_code == 200, response.text + body = response.json() + try: + assert body.get("indexed_count", 0) > 0, f"nothing was indexed: {body}" + assert body.get("failed_count", 1) == 0, f"a chunk failed to index: {body}" + assert FILENAME in (body.get("uploaded") or []), body + + listed = client.get(PERSONAL_PATH) + assert listed.status_code == 200, listed.text + names = [str(f.get("name")) for f in listed.json().get("files") or []] + assert any(STEM in name for name in names), names + finally: + listed = client.get(PERSONAL_PATH).json().get("files") or [] + for entry in listed: + if STEM in str(entry.get("name")): + client.request("DELETE", "/api/personal/file", + params={"filepath": entry.get("path")}) diff --git a/tests/smoke/test_documents_smoke.py b/tests/smoke/test_documents_smoke.py new file mode 100644 index 000000000..978571f12 --- /dev/null +++ b/tests/smoke/test_documents_smoke.py @@ -0,0 +1,46 @@ +"""Documents: the editor's create, edit and version history survive a round trip.""" +from __future__ import annotations + +DOCUMENT_PATH = "/api/document" +LIBRARY_PATH = "/api/documents/library" + +TITLE = "Odysseus smoke document" +FIRST = "First revision, written by the release smoke suite." +SECOND = "Second revision, written by the release smoke suite." + + +def test_a_document_round_trips_with_its_versions(client): + created = client.post(DOCUMENT_PATH, json={"title": TITLE, "content": FIRST}) + assert created.status_code == 200, created.text + body = created.json() + doc_id = body["id"] + try: + assert body.get("current_content") == FIRST, body + assert body.get("version_count") == 1, body + + library = client.get(LIBRARY_PATH) + assert library.status_code == 200, library.text + assert doc_id in [d.get("id") for d in library.json().get("documents") or []] + + # `force_version` because a save inside the route's coalesce + # window updates the current version in place instead of adding + # one - which is right for autosave and would make a smoke check + # that edits immediately depend on the clock. + edited = client.put(f"{DOCUMENT_PATH}/{doc_id}", + json={"content": SECOND, "force_version": True}) + assert edited.status_code == 200, edited.text + assert edited.json().get("current_content") == SECOND, edited.text + assert edited.json().get("version_count") == 2, edited.text + + versions = client.get(f"{DOCUMENT_PATH}/{doc_id}/versions") + assert versions.status_code == 200, versions.text + contents = {v.get("version_number"): v.get("content") for v in versions.json()} + assert contents.get(1) == FIRST, contents + assert contents.get(2) == SECOND, contents + + restored = client.post(f"{DOCUMENT_PATH}/{doc_id}/restore/1") + assert restored.status_code == 200, restored.text + assert client.get(f"{DOCUMENT_PATH}/{doc_id}").json()["current_content"] == FIRST + finally: + removed = client.delete(f"{DOCUMENT_PATH}/{doc_id}") + assert removed.status_code == 200, removed.text diff --git a/tests/smoke/test_email_smoke.py b/tests/smoke/test_email_smoke.py new file mode 100644 index 000000000..32665225f --- /dev/null +++ b/tests/smoke/test_email_smoke.py @@ -0,0 +1,96 @@ +"""Email: the inbox lists a seeded message, opens it, and marks it read. + +Email is the one area with no way to reach a real account deterministically, +and the repo already solved that: `routes/email_routes.py` carries a +fixture path gated on `ODYSSEUS_EMAIL_FIXTURE=1` plus a fixture file in +the data dir. This uses that mechanism rather than inventing a second +one - which means it also only covers what the fixture covers. Real IMAP +sync and SMTP send stay out, and say so in the table's gap list. +""" +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from src.constants import DATA_DIR + +LIST_PATH = "/api/email/list" +READ_PATH = "/api/email/read" +MARK_READ_PATH = "/api/email/mark-read" +UNREAD_STATE_PATH = "/api/email/unread-state" + +# The filename the fixture path reads. Same value as +# routes/email_routes.py's `_fixture_email_file`. +FIXTURE_FILENAME = "fixture_email_messages.json" + +SUBJECT = "Odysseus smoke inbox message" +BODY = "Body of the smoke fixture message." +SENDER = "Smoke Sender " + + +@pytest.fixture +def seeded_inbox(client, account): + """Write the fixture inbox, and put back whatever was there before. + + The flag itself has to be in the app's environment, which is the + launcher's job; if it is missing the fixture path stays off and the + list route falls through to a real account that does not exist. That + reads as a skip, not a failure. + """ + path = Path(DATA_DIR) / FIXTURE_FILENAME + previous = path.read_bytes() if path.exists() else None + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps({"messages": [{ + "owner": account["username"], + "from": SENDER, + "subject": SUBJECT, + "date": "2026-09-29T12:00:00+00:00", + "body": BODY, + }]}, indent=2) + "\n", encoding="utf-8") + try: + yield path + finally: + if previous is None: + path.unlink(missing_ok=True) + else: + path.write_bytes(previous) + + +def _fixture_rows(client): + response = client.get(LIST_PATH, params={"folder": "INBOX", "limit": 10}) + assert response.status_code == 200, response.text + body = response.json() + rows = [e for e in body.get("emails") or [] if e.get("subject") == SUBJECT] + if not rows: + pytest.skip( + "the instance is not serving the email fixture, so there is no " + "deterministic inbox to read. Boot it with ODYSSEUS_EMAIL_FIXTURE=1 " + "(scripts/odysseus-smoke does)." + ) + return rows + + +def test_the_inbox_lists_opens_and_marks_a_message(client, seeded_inbox): + row = _fixture_rows(client)[0] + uid = row["uid"] + assert row.get("from_address") == "smoke@example.invalid", row + assert row.get("is_read") is False, row + + read = client.get(f"{READ_PATH}/{uid}", params={"folder": "INBOX"}) + assert read.status_code == 200, read.text + opened = read.json() + assert opened.get("subject") == SUBJECT, opened + assert BODY in str(opened.get("body") or ""), opened + assert BODY in str(opened.get("body_html") or ""), opened + + before = client.get(UNREAD_STATE_PATH, params={"folder": "INBOX"}) + assert before.status_code == 200, before.text + assert before.json().get("unread_count") == 1, before.text + + marked = client.post(f"{MARK_READ_PATH}/{uid}", params={"folder": "INBOX"}) + assert marked.status_code == 200, marked.text + + after = client.get(UNREAD_STATE_PATH, params={"folder": "INBOX"}) + assert after.json().get("unread_count") == 0, after.text diff --git a/tests/smoke/test_memory_smoke.py b/tests/smoke/test_memory_smoke.py new file mode 100644 index 000000000..e0c03a4eb --- /dev/null +++ b/tests/smoke/test_memory_smoke.py @@ -0,0 +1,40 @@ +"""Memory: a stored fact is listed, found by search, and gone after delete. + +Keyword mode is enough here on purpose. The memory store degrades to +keyword matching when no vector service answers, and that degraded path +is the one a clean checkout actually runs, so it is the one worth +smoking. +""" +from __future__ import annotations + +MEMORY_PATH = "/api/memory" +ADD_PATH = "/api/memory/add" +SEARCH_PATH = "/api/memory/search" + +# A token that cannot collide with a real memory on a scratch instance. +TOKEN = "odysseus-smoke-marker-quintile" +TEXT = f"The release smoke suite stored the token {TOKEN} as a fact." + + +def test_a_memory_round_trips(client): + created = client.post(ADD_PATH, json={"text": TEXT, "category": "fact"}) + assert created.status_code == 200, created.text + assert created.json().get("ok") is True, created.text + + listed = client.get(MEMORY_PATH) + assert listed.status_code == 200, listed.text + matching = [m for m in listed.json().get("memory") or [] if TOKEN in str(m.get("text"))] + assert matching, [m.get("text") for m in listed.json().get("memory") or []] + memory_id = matching[0]["id"] + + try: + found = client.post(SEARCH_PATH, data={"query": TOKEN}) + assert found.status_code == 200, found.text + hits = [m for m in found.json().get("memories") or [] if TOKEN in str(m.get("text"))] + assert hits, found.text + finally: + removed = client.delete(f"{MEMORY_PATH}/{memory_id}") + assert removed.status_code == 200, removed.text + + remaining = client.get(MEMORY_PATH).json().get("memory") or [] + assert memory_id not in [m.get("id") for m in remaining] diff --git a/tests/smoke/test_notes_smoke.py b/tests/smoke/test_notes_smoke.py new file mode 100644 index 000000000..4b4381144 --- /dev/null +++ b/tests/smoke/test_notes_smoke.py @@ -0,0 +1,32 @@ +"""Notes: a note created through the API is readable, editable and gone after delete.""" +from __future__ import annotations + +NOTES_PATH = "/api/notes" + +TITLE = "Odysseus smoke note" +BODY = "Created by the release smoke suite." +EDITED_BODY = "Edited by the release smoke suite." + + +def test_a_note_round_trips(client): + created = client.post(NOTES_PATH, json={"title": TITLE, "content": BODY}) + assert created.status_code == 200, created.text + note_id = created.json()["id"] + try: + listed = client.get(NOTES_PATH) + assert listed.status_code == 200, listed.text + titles = [n.get("title") for n in listed.json().get("notes") or []] + assert TITLE in titles, titles + + read = client.get(f"{NOTES_PATH}/{note_id}") + assert read.status_code == 200, read.text + assert read.json().get("content") == BODY, read.text + + edited = client.put(f"{NOTES_PATH}/{note_id}", + json={"title": TITLE, "content": EDITED_BODY}) + assert edited.status_code == 200, edited.text + assert client.get(f"{NOTES_PATH}/{note_id}").json()["content"] == EDITED_BODY + finally: + removed = client.delete(f"{NOTES_PATH}/{note_id}") + assert removed.status_code == 200, removed.text + assert client.get(f"{NOTES_PATH}/{note_id}").status_code == 404 diff --git a/tests/smoke/test_settings_smoke.py b/tests/smoke/test_settings_smoke.py new file mode 100644 index 000000000..4ebcd3cdf --- /dev/null +++ b/tests/smoke/test_settings_smoke.py @@ -0,0 +1,31 @@ +"""Settings: a preference written through the API survives a new login. + +Reading the value back on the same cookie only proves the handler +answered. Reading it back after authenticating again is what proves it +was persisted rather than held in the session, which is the closest an +API-level check gets to the user reloading the page. +""" +from __future__ import annotations + +PREFS_PATH = "/api/prefs" + +KEY = "odysseus_smoke_preference" +VALUE = "set-by-the-release-smoke-suite" + + +def test_a_preference_survives_a_new_login(client, fresh_client): + written = client.put(f"{PREFS_PATH}/{KEY}", json={"value": VALUE}) + assert written.status_code == 200, written.text + assert written.json().get("value") == VALUE, written.text + + read = client.get(f"{PREFS_PATH}/{KEY}") + assert read.status_code == 200, read.text + assert read.json().get("value") == VALUE, read.text + + reloaded = fresh_client.get(f"{PREFS_PATH}/{KEY}") + assert reloaded.status_code == 200, reloaded.text + assert reloaded.json().get("value") == VALUE, reloaded.text + + listed = fresh_client.get(PREFS_PATH) + assert listed.status_code == 200, listed.text + assert listed.json().get(KEY) == VALUE, listed.text diff --git a/tests/smoke/test_tasks_smoke.py b/tests/smoke/test_tasks_smoke.py new file mode 100644 index 000000000..8f6eff5f1 --- /dev/null +++ b/tests/smoke/test_tasks_smoke.py @@ -0,0 +1,41 @@ +"""Tasks: a scheduled task is created with a computed next run and is listed. + +Deliberately not fired. Running a task is model and tool work the +checkpoint benchmark covers; what this asserts is that the scheduler +still accepts a task and computes when it should run, which is the part +a route move can break silently. +""" +from __future__ import annotations + +TASKS_PATH = "/api/tasks" + +NAME = "Odysseus smoke task" +SCHEDULED_TIME = "03:00" + + +def test_a_scheduled_task_round_trips(client): + created = client.post(TASKS_PATH, json={ + "name": NAME, + "task_type": "llm", + "prompt": "Smoke task; never run by this suite.", + "trigger_type": "schedule", + "schedule": "daily", + "scheduled_time": SCHEDULED_TIME, + }) + assert created.status_code == 200, created.text + body = created.json() + task_id = body["id"] + try: + assert body.get("next_run"), f"no next run computed for a daily task: {body}" + assert body.get("status") == "active", body + + listed = client.get(TASKS_PATH) + assert listed.status_code == 200, listed.text + assert task_id in [t.get("id") for t in listed.json().get("tasks") or []] + + paused = client.post(f"{TASKS_PATH}/{task_id}/pause") + assert paused.status_code == 200, paused.text + assert client.get(f"{TASKS_PATH}/{task_id}").json().get("status") == "paused" + finally: + removed = client.delete(f"{TASKS_PATH}/{task_id}") + assert removed.status_code == 200, removed.text diff --git a/tests/smoke/test_uploads_smoke.py b/tests/smoke/test_uploads_smoke.py new file mode 100644 index 000000000..c5527ccbf --- /dev/null +++ b/tests/smoke/test_uploads_smoke.py @@ -0,0 +1,27 @@ +"""Uploads: a file uploaded through the chat attachment route reads back byte for byte.""" +from __future__ import annotations + +UPLOAD_PATH = "/api/upload" +STATS_PATH = "/api/upload/stats" + +FILENAME = "odysseus-smoke-attachment.txt" +CONTENT = b"Uploaded by the release smoke suite." + + +def test_an_upload_reads_back_unchanged(client): + response = client.post(UPLOAD_PATH, + files={"files": (FILENAME, CONTENT, "text/plain")}) + assert response.status_code == 200, response.text + files = response.json().get("files") or [] + assert len(files) == 1, response.text + entry = files[0] + assert entry.get("name") == FILENAME, entry + assert entry.get("size") == len(CONTENT), entry + + fetched = client.get(f"{UPLOAD_PATH}/{entry['id']}") + assert fetched.status_code == 200, fetched.text + assert fetched.content == CONTENT, fetched.content + + stats = client.get(STATS_PATH) + assert stats.status_code == 200, stats.text + assert stats.json().get("total_files", 0) >= 1, stats.text diff --git a/tests/test_smoke_area_table.py b/tests/test_smoke_area_table.py new file mode 100644 index 000000000..8827fa223 --- /dev/null +++ b/tests/test_smoke_area_table.py @@ -0,0 +1,96 @@ +"""The release smoke suite's report cannot overstate what it checked. + +The suite's value is entirely in whether its table is honest, and the +table is built from a registry rather than from what happened to run. +These pin the properties that make it honest: every advertised area has +a row whether or not its module ran, every module that exists is +registered, a failure outranks a pass, and the whole thing stays ASCII. +""" +from pathlib import Path + +import pytest + +from tests.smoke import areas + +SMOKE_DIR = Path(__file__).parent / "smoke" + + +def test_every_registered_area_has_its_module_on_disk(): + missing = [area.module for area in areas.COVERED + if not (SMOKE_DIR / area.module).exists()] + assert not missing, f"registered areas with no test module: {missing}" + + +def test_every_smoke_module_is_registered(): + """A module nobody registered would run and never appear in the table.""" + on_disk = {path.name for path in SMOKE_DIR.glob("test_*_smoke.py")} + registered = {area.module for area in areas.COVERED} + assert on_disk == registered, ( + f"unregistered modules: {sorted(on_disk - registered)}; " + f"registered but absent: {sorted(registered - on_disk)}" + ) + + +def test_area_keys_are_unique(): + keys = [area.key for area in areas.COVERED] + assert len(keys) == len(set(keys)), keys + + +def test_a_run_that_reported_nothing_shows_every_area_as_not_run(): + table = areas.render_table({}) + for area in areas.COVERED: + assert area.label in table, area.label + assert table.count(areas.NOT_RUN) == len(areas.COVERED) + assert areas.PASS not in table + + +def test_an_unreported_area_is_not_dropped_from_the_table(): + """The registry decides the rows, so a partial run still lists the rest.""" + table = areas.render_table({areas.COVERED[0].key: {"result": areas.PASS, "checks": 1}}) + assert table.count(areas.NOT_RUN) == len(areas.COVERED) - 1 + assert areas.COVERED[-1].label in table + + +def test_declared_gaps_are_printed_with_their_reason(): + assert areas.DECLARED_GAPS, "a suite with no declared gaps is claiming total coverage" + table = areas.render_table({}) + for gap in areas.DECLARED_GAPS: + assert gap.label in table, gap.label + # The reason is wrapped across lines, so match its first words. + assert " ".join(gap.reason.split()[:3]) in " ".join(table.split()), gap.reason + + +@pytest.mark.parametrize("outcomes,expected", [ + ([], areas.NOT_RUN), + ([areas.PASS], areas.PASS), + ([areas.PASS, areas.SKIP], areas.SKIP), + ([areas.PASS, areas.SKIP, areas.FAIL], areas.FAIL), + ([areas.PASS, areas.FAIL], areas.FAIL), +]) +def test_the_worst_outcome_decides_the_row(outcomes, expected): + assert areas.resolve(outcomes) == expected + + +def test_the_table_is_ascii_only(): + """No emoji or status glyphs: the repo bans them in UI and in code.""" + table = areas.render_table({area.key: {"result": areas.PASS, "checks": 1} + for area in areas.COVERED}) + assert table.isascii(), [ch for ch in table if not ch.isascii()] + + +def test_module_names_map_back_to_their_area(): + for area in areas.COVERED: + assert areas.area_for_module(area.module) == area.key + assert areas.area_for_module("test_not_a_smoke_module.py") is None + + +def test_the_summary_line_counts_every_row(): + results = {area.key: {"result": areas.PASS, "checks": 1} for area in areas.COVERED} + results[areas.COVERED[0].key] = {"result": areas.FAIL, "checks": 0} + results[areas.COVERED[1].key] = {"result": areas.SKIP, "checks": 0} + summary = areas.render_table(results).splitlines()[-1] + assert f"{len(areas.COVERED) - 2} pass" in summary, summary + assert "1 fail" in summary, summary + assert "1 skip" in summary, summary + assert "0 not run" in summary, summary + assert f"{len(areas.DECLARED_GAPS)} declared gaps" in summary, summary