mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-12 01:42:20 +02:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6ee6502010 |
@@ -30,6 +30,8 @@ secrets.env~
|
||||
.idea/
|
||||
dev-docs/
|
||||
docs/
|
||||
website/
|
||||
assets/branding/
|
||||
*.md
|
||||
*.db
|
||||
*.sqlite
|
||||
|
||||
+117
-2
@@ -67,6 +67,11 @@ SEARXNG_INSTANCE=http://localhost:8080
|
||||
# Auth & Security
|
||||
# ============================================================
|
||||
|
||||
# Optional backend workspace used automatically by the WebUI when no workspace
|
||||
# is saved in the browser. This must be a directory visible to the backend;
|
||||
# with host-workspace mapping, a host path is translated before vetting.
|
||||
# ODYSSEUS_WORKSPACE_DEFAULT=/workspace/project
|
||||
|
||||
# Enable authentication (default: true)
|
||||
# AUTH_ENABLED=true
|
||||
|
||||
@@ -76,12 +81,32 @@ SEARXNG_INSTANCE=http://localhost:8080
|
||||
# Change this if another local service already uses 7000 (macOS AirPlay often does).
|
||||
# APP_PORT=7000
|
||||
|
||||
# Optional HTTP address advertised in companion/mobile pairing codes. Set this
|
||||
# when Docker would otherwise advertise a container address or loopback. Use a
|
||||
# LAN or Tailscale IPv4 address, a single-label hostname, or an mDNS *.local
|
||||
# name that the phone can reach. HTTPS and public hostnames are not supported
|
||||
# by the current companion client. Do not include credentials, a path, query,
|
||||
# or fragment.
|
||||
# COMPANION_BASE_URL=http://192.168.1.50:7000
|
||||
|
||||
# Development-only auth bypass for loopback requests.
|
||||
# Keep false for Docker, LAN, reverse proxy, and any shared deployment.
|
||||
# LOCALHOST_BYPASS=false
|
||||
|
||||
# Mark session cookies Secure. Set true when Odysseus is served through HTTPS
|
||||
# by a trusted reverse proxy or private access gateway.
|
||||
# Skip the external-context exact-approval pause for unattended local agents.
|
||||
# Keep false for shared or internet-exposed deployments.
|
||||
# ODYSSEUS_UNATTENDED_MODE=false
|
||||
|
||||
# Optional post-external-context tool approval gate. Off by default because it
|
||||
# can block normal agent work; enable only for deployments that want this fence.
|
||||
# ODYSSEUS_TOOL_APPROVAL_GATE=0
|
||||
|
||||
# Mark session cookies Secure. Left unset, this follows the request scheme:
|
||||
# an HTTPS login gets a Secure cookie, a plain-HTTP one does not. Set true to
|
||||
# force it on, or false to force it off while you still serve plain HTTP.
|
||||
# Upgrading: this used to default to false. Drop a leftover SECURE_COOKIES=false
|
||||
# from your .env unless you still need that escape hatch — it keeps HTTPS logins
|
||||
# on a non-Secure cookie.
|
||||
# SECURE_COOKIES=true
|
||||
|
||||
# Optional: pre-seed the first admin password during setup.
|
||||
@@ -130,6 +155,42 @@ SEARXNG_INSTANCE=http://localhost:8080
|
||||
# FASTEMBED_MODEL=sentence-transformers/all-MiniLM-L6-v2
|
||||
# FASTEMBED_CACHE_PATH= # defaults to ~/.cache/fastembed
|
||||
|
||||
# ============================================================
|
||||
# Google OAuth2 (Google Workspace / .edu email accounts)
|
||||
# ============================================================
|
||||
# Required to use the "Connect with Google" OAuth flow in email account setup.
|
||||
# Create credentials at: console.cloud.google.com → APIs & Services → Credentials
|
||||
# 1. Enable the Gmail API for your project.
|
||||
# 2. Configure the OAuth consent screen (User Type: Internal for Workspace orgs).
|
||||
# Add scopes: https://mail.google.com/ and email.
|
||||
# 3. Create an OAuth 2.0 Client ID (type: Web application).
|
||||
# Add your redirect URI: http://localhost:7000/api/email/oauth/google/callback
|
||||
# (replace host/port for hosted installs).
|
||||
# 4. Copy the Client ID and Client Secret below.
|
||||
#
|
||||
# GOOGLE_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
|
||||
# GOOGLE_OAUTH_CLIENT_SECRET=replace-with-client-secret
|
||||
#
|
||||
# Set this explicitly for HTTPS, reverse-proxy, or hosted deployments. The
|
||||
# value must exactly match an authorized redirect URI in the Google client.
|
||||
# Local HTTP setups may use the callback URL inferred by the application.
|
||||
# GOOGLE_OAUTH_REDIRECT_URI=https://your-domain.com/api/email/oauth/google/callback
|
||||
|
||||
# Origin the MCP OAuth callback is sent back to, for remote (Streamable HTTP)
|
||||
# MCP servers that register it dynamically. Defaults to http://localhost:$APP_PORT,
|
||||
# which is right only when you reach Odysseus directly on that port. Set it for
|
||||
# HTTPS, reverse-proxy, hosted, and Docker installs — inside the container the
|
||||
# app always listens on 7000 and cannot see the host port map, so the default is
|
||||
# wrong there whenever APP_PORT is not 7000.
|
||||
#
|
||||
# Not for Google MCP servers. Those use Desktop App credentials, and Google only
|
||||
# accepts loopback redirect URIs for that client type, so a public origin here is
|
||||
# rejected with redirect_uri_mismatch. Leave it unset for a Google-only install:
|
||||
# the loopback default is what Google wants, and remote users finish through the
|
||||
# paste-back page, which never has to load the redirect.
|
||||
# https://developers.google.com/identity/protocols/oauth2/native-app
|
||||
# OAUTH_REDIRECT_BASE_URL=https://your-domain.com
|
||||
|
||||
# ============================================================
|
||||
# Misc
|
||||
# ============================================================
|
||||
@@ -168,6 +229,58 @@ SEARXNG_INSTANCE=http://localhost:8080
|
||||
# ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES=26214400 # email compose attachment (25 MB)
|
||||
# ODYSSEUS_STT_MAX_AUDIO_BYTES=26214400 # speech-to-text audio (25 MB)
|
||||
# ODYSSEUS_ICS_MAX_BYTES=10485760 # calendar .ics import (10 MB)
|
||||
# ODYSSEUS_TTS_CACHE_MAX_BYTES=524288000 # TTS cache (500 MB)
|
||||
|
||||
# ============================================================
|
||||
# Host Docker access (explicit opt-in)
|
||||
# ============================================================
|
||||
# Default Docker Compose does not mount /var/run/docker.sock. Existing
|
||||
# Ollama, vLLM, and other OpenAI-compatible endpoints remain usable without it.
|
||||
#
|
||||
# Enable this only for intentional Cookbook/local Docker-daemon management.
|
||||
# Raw socket access is high-trust and can grant broad control over the host
|
||||
# Docker daemon. Set DOCKER_GID to the host docker group's numeric GID.
|
||||
# Put these values in .env, or export them before running docker compose.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-docker.yml
|
||||
# DOCKER_GID=963
|
||||
# docker/host-docker.yml sets this inside the container. Keep it paired
|
||||
# with the socket overlay; setting it alone is not sufficient.
|
||||
# ODYSSEUS_ENABLE_HOST_DOCKER=true
|
||||
#
|
||||
# Host Docker access can be combined with one GPU overlay:
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/gpu.nvidia.yml:docker/host-docker.yml
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/gpu.amd.yml:docker/host-docker.yml
|
||||
|
||||
# ============================================================
|
||||
# Host workspace access (explicit opt-in)
|
||||
# ============================================================
|
||||
# Docker installs normally see only the container filesystem and /app/data.
|
||||
# Enable this when the agent should edit a real host workspace like Codex.
|
||||
# This is high-trust: the mounted tree is writable by the Odysseus container.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml
|
||||
# ODYSSEUS_HOST_WORKSPACE_DIR=/home/you
|
||||
# ODYSSEUS_HOST_WORKSPACE_MOUNT=/host/workspace
|
||||
#
|
||||
# Host workspace access can be combined with host Docker access and GPU overlays:
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml:docker/host-docker.yml
|
||||
|
||||
# ============================================================
|
||||
# Host network access (explicit opt-in, Linux Docker)
|
||||
# ============================================================
|
||||
# Docker bridge networking hides some host/LAN/VPN behavior from the agent:
|
||||
# mDNS, some LAN discovery, local VPN/Tailscale state, and host namespace
|
||||
# assumptions may differ from native Codex. Enable this only for high-trust
|
||||
# local installs where the Odysseus container should share the host network.
|
||||
#
|
||||
# With host networking, Docker port publishing is disabled and the app listens
|
||||
# directly on APP_PORT. The bundled SearXNG/Chroma services stay in Docker and
|
||||
# are reached through their host-published loopback ports.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml:docker/host-network.yml
|
||||
# APP_BIND=127.0.0.1
|
||||
# APP_PORT=7000
|
||||
# ODYSSEUS_HOST_NETWORK_SEARXNG_INSTANCE=http://127.0.0.1:8080
|
||||
# ODYSSEUS_HOST_NETWORK_CHROMADB_HOST=127.0.0.1
|
||||
# ODYSSEUS_HOST_NETWORK_CHROMADB_PORT=8100
|
||||
|
||||
# ============================================================
|
||||
# GPU support (Docker Compose)
|
||||
@@ -197,3 +310,5 @@ SEARXNG_INSTANCE=http://localhost:8080
|
||||
|
||||
# APP_DATA_DIR=./data
|
||||
# APP_LOGS_DIR=./logs
|
||||
# Maximum serialized layered photo-editor draft size (default: 256 MiB).
|
||||
ODYSSEUS_EDITOR_DRAFT_MAX_BYTES=268435456
|
||||
|
||||
@@ -15,6 +15,13 @@ docker/entrypoint.sh text eol=lf
|
||||
*.cmd text eol=crlf
|
||||
*.bat text eol=crlf
|
||||
|
||||
# Vendored third-party bundles in static/lib/ are published minified artifacts
|
||||
# and must stay byte-identical to what npm ships — stripping trailing whitespace
|
||||
# to satisfy `git diff --check` would desync them from the upstream release. Turn
|
||||
# the whitespace check off for that tree instead, and keep the bundles out of
|
||||
# GitHub's language statistics.
|
||||
static/lib/** -whitespace linguist-vendored
|
||||
|
||||
# Binary assets — never normalize.
|
||||
*.png binary
|
||||
*.jpg binary
|
||||
|
||||
+1
-1
@@ -6,4 +6,4 @@
|
||||
# A per-area ownership map (security/auth, CI, frontend, agent internals, with
|
||||
# multiple named owners per line) is being worked out in issue #593; once
|
||||
# agreed it replaces this file. Until then, required reviews and the security
|
||||
# CI gate (docs/security-ci.md) remain in force via branch protection.
|
||||
# CI gate (website/security-ci.md) remain in force via branch protection.
|
||||
|
||||
@@ -6,26 +6,38 @@ body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
**Before submitting:** search [open issues](https://github.com/pewdiepie-archdaemon/odysseus/issues)
|
||||
and [discussions](https://github.com/pewdiepie-archdaemon/odysseus/discussions) first.
|
||||
**Before submitting:** search [open issues](https://github.com/odysseus-dev/odysseus/issues)
|
||||
and [discussions](https://github.com/odysseus-dev/odysseus/discussions) first.
|
||||
Duplicate reports slow things down.
|
||||
|
||||
For security vulnerabilities, **do not open a public issue** —
|
||||
use [GitHub Security Advisories](https://github.com/pewdiepie-archdaemon/odysseus/security/advisories/new)
|
||||
and read [SECURITY.md](https://github.com/pewdiepie-archdaemon/odysseus/blob/main/SECURITY.md) first.
|
||||
use [GitHub Security Advisories](https://github.com/odysseus-dev/odysseus/security/advisories/new)
|
||||
and read [SECURITY.md](https://github.com/odysseus-dev/odysseus/blob/main/SECURITY.md) first.
|
||||
|
||||
- type: checkboxes
|
||||
id: prerequisites
|
||||
attributes:
|
||||
label: Prerequisites
|
||||
options:
|
||||
- label: I searched [open issues](https://github.com/pewdiepie-archdaemon/odysseus/issues?q=is%3Aissue+is%3Aopen) and [discussions](https://github.com/pewdiepie-archdaemon/odysseus/discussions) and did not find an existing report of this bug.
|
||||
- label: I searched [open issues](https://github.com/odysseus-dev/odysseus/issues?q=is%3Aissue+is%3Aopen) and [discussions](https://github.com/odysseus-dev/odysseus/discussions) and did not find an existing report of this bug.
|
||||
required: true
|
||||
- label: This is **not** a security vulnerability. (Vulnerabilities go to [GitHub Security Advisories](https://github.com/pewdiepie-archdaemon/odysseus/security/advisories/new) — see [SECURITY.md](https://github.com/pewdiepie-archdaemon/odysseus/blob/main/SECURITY.md).)
|
||||
- label: This is **not** a security vulnerability. (Vulnerabilities go to [GitHub Security Advisories](https://github.com/odysseus-dev/odysseus/security/advisories/new) — see [SECURITY.md](https://github.com/odysseus-dev/odysseus/blob/main/SECURITY.md).)
|
||||
required: true
|
||||
- label: I am running the latest code from the `dev` branch (the default branch you get on clone, where fixes land first) and the bug still reproduces there. Please `git pull` the latest `dev` before filing.
|
||||
required: true
|
||||
|
||||
- type: input
|
||||
id: revision
|
||||
attributes:
|
||||
label: Odysseus Revision
|
||||
description: |
|
||||
From the repository root (on the host when using Docker), run
|
||||
`git show -s --abbrev=12 --format='%h (%cs)' HEAD`
|
||||
and paste the output exactly.
|
||||
placeholder: "1fef4929cf1d (2026-08-11)"
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: dropdown
|
||||
id: install-method
|
||||
attributes:
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
blank_issues_enabled: false
|
||||
contact_links:
|
||||
- name: Question / Need Help
|
||||
url: https://github.com/pewdiepie-archdaemon/odysseus/discussions/categories/q-a
|
||||
url: https://github.com/odysseus-dev/odysseus/discussions/categories/q-a
|
||||
about: Ask how-to questions, setup help, and model configuration questions here. Issues are for confirmed bugs and concrete proposals only.
|
||||
|
||||
- name: Idea or Suggestion
|
||||
url: https://github.com/pewdiepie-archdaemon/odysseus/discussions/categories/ideas
|
||||
url: https://github.com/odysseus-dev/odysseus/discussions/categories/ideas
|
||||
about: Discuss ideas and gauge interest before opening a formal feature request. If there is already a discussion, link it in your feature request.
|
||||
|
||||
- name: Security Vulnerability
|
||||
url: https://github.com/pewdiepie-archdaemon/odysseus/security/advisories/new
|
||||
url: https://github.com/odysseus-dev/odysseus/security/advisories/new
|
||||
about: Report vulnerabilities privately via GitHub Security Advisories — never as a public issue. Read SECURITY.md before reporting.
|
||||
|
||||
@@ -6,22 +6,22 @@ body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
**Before submitting:** search [open issues](https://github.com/pewdiepie-archdaemon/odysseus/issues)
|
||||
and [discussions](https://github.com/pewdiepie-archdaemon/odysseus/discussions) first.
|
||||
Feature requests that duplicate [ROADMAP.md](https://github.com/pewdiepie-archdaemon/odysseus/blob/main/ROADMAP.md)
|
||||
**Before submitting:** search [open issues](https://github.com/odysseus-dev/odysseus/issues)
|
||||
and [discussions](https://github.com/odysseus-dev/odysseus/discussions) first.
|
||||
Feature requests that duplicate [ROADMAP.md](https://github.com/odysseus-dev/odysseus/blob/main/ROADMAP.md)
|
||||
or an existing open issue will be closed as duplicates.
|
||||
|
||||
If your idea needs community input before it becomes a concrete proposal,
|
||||
start a [discussion](https://github.com/pewdiepie-archdaemon/odysseus/discussions/categories/ideas) instead.
|
||||
start a [discussion](https://github.com/odysseus-dev/odysseus/discussions/categories/ideas) instead.
|
||||
|
||||
- type: checkboxes
|
||||
id: prerequisites
|
||||
attributes:
|
||||
label: Prerequisites
|
||||
options:
|
||||
- label: I searched [open issues](https://github.com/pewdiepie-archdaemon/odysseus/issues?q=is%3Aissue+is%3Aopen) and this has not already been proposed.
|
||||
- label: I searched [open issues](https://github.com/odysseus-dev/odysseus/issues?q=is%3Aissue+is%3Aopen) and this has not already been proposed.
|
||||
required: true
|
||||
- label: I searched [discussions](https://github.com/pewdiepie-archdaemon/odysseus/discussions) and this is not already being debated there.
|
||||
- label: I searched [discussions](https://github.com/odysseus-dev/odysseus/discussions) and this is not already being debated there.
|
||||
required: true
|
||||
- label: This is a concrete, actionable proposal — not a vague "it would be nice if..." request.
|
||||
required: true
|
||||
|
||||
@@ -24,10 +24,11 @@ Fixes #
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] I searched [open issues](https://github.com/pewdiepie-archdaemon/odysseus/issues) and [open PRs](https://github.com/pewdiepie-archdaemon/odysseus/pulls) — this is not a duplicate.
|
||||
- [ ] I searched [open issues](https://github.com/odysseus-dev/odysseus/issues) and [open PRs](https://github.com/odysseus-dev/odysseus/pulls) — this is not a duplicate.
|
||||
- [ ] This PR targets `dev`
|
||||
- [ ] My changes are limited to the scope described above — no unrelated refactors or whitespace changes mixed in.
|
||||
- [ ] I actually ran the app (`docker compose up` or `uvicorn app:app`) and verified the change works end-to-end. Type-checks and unit tests are not enough.
|
||||
- [ ] I did not run the app/runtime validation and stated that gap in **How to Test**. Leave this unchecked when the app-run box above is checked.
|
||||
|
||||
## How to Test
|
||||
|
||||
|
||||
@@ -41,6 +41,14 @@ module.exports = async ({ github, context, core }) => {
|
||||
break;
|
||||
|
||||
case 'bug': {
|
||||
const revisionText = section('Odysseus Revision');
|
||||
if (!/^[0-9a-f]{12} \(\d{4}-\d{2}-\d{2}\)$/i.test(revisionText)) {
|
||||
failures.push(
|
||||
'**Odysseus Revision** — paste the 12-character commit SHA and date, ' +
|
||||
'for example `1fef4929cf1d (2026-08-11)`',
|
||||
);
|
||||
}
|
||||
|
||||
if (!section('Install Method')) {
|
||||
failures.push('**Install Method** — select how you installed Odysseus');
|
||||
}
|
||||
@@ -153,6 +161,16 @@ module.exports = async ({ github, context, core }) => {
|
||||
}
|
||||
}
|
||||
|
||||
const LABEL_BAD = 'needs more info';
|
||||
const LABEL_GOOD = 'ready for review';
|
||||
|
||||
// Closed issues are no longer awaiting review.
|
||||
// This also prevents later edits to closed issues from restoring the label.
|
||||
if (issue.state === 'closed') {
|
||||
await dropLabel(LABEL_GOOD);
|
||||
return;
|
||||
}
|
||||
|
||||
// ── Find existing bot comment to update in-place ──────────────────────────
|
||||
const MARKER = '<!-- issue-description-check -->';
|
||||
const { data: comments } = await github.rest.issues.listComments({
|
||||
@@ -160,9 +178,6 @@ module.exports = async ({ github, context, core }) => {
|
||||
});
|
||||
const existing = comments.find(c => c.user.type === 'Bot' && c.body.includes(MARKER));
|
||||
|
||||
const LABEL_BAD = 'needs more info';
|
||||
const LABEL_GOOD = 'ready for review';
|
||||
|
||||
if (failures.length === 0) {
|
||||
if (existing) {
|
||||
await github.rest.issues.deleteComment({ owner, repo, comment_id: existing.id });
|
||||
|
||||
@@ -21,11 +21,11 @@ module.exports = async ({ github, context, core }) => {
|
||||
return strip(m?.[0].replace(new RegExp(`#+\\s+${heading}`, 'i'), '') ?? '');
|
||||
}
|
||||
|
||||
const problems = [];
|
||||
const descriptionProblems = [];
|
||||
|
||||
// 1. Summary must be filled in.
|
||||
if (section('Summary').length < 20) {
|
||||
problems.push('**Summary** is empty or too short — describe what changed and why.');
|
||||
descriptionProblems.push('**Summary** is empty or too short — describe what changed and why.');
|
||||
}
|
||||
|
||||
// 2. Linked Issue must reference a real issue. Accept a bare #NNN, a closing
|
||||
@@ -34,18 +34,18 @@ module.exports = async ({ github, context, core }) => {
|
||||
const linkedSection = section('Linked Issue');
|
||||
const hasIssueRef = /#\d+\b/.test(linkedSection) || /\/issues\/\d+/.test(linkedSection);
|
||||
if (!linkedSection || !hasIssueRef) {
|
||||
problems.push('**Linked Issue** — add a reference like `Fixes #NNN`, a bare `#NNN`, or a link to the issue.');
|
||||
descriptionProblems.push('**Linked Issue** — add a reference like `Fixes #NNN`, a bare `#NNN`, or a link to the issue.');
|
||||
}
|
||||
|
||||
// 3. At least one Type of Change box must be checked.
|
||||
const typeBlock = body.match(/##\s+Type of Change[\s\S]*?(?=\n##\s|$)/i)?.[0] ?? '';
|
||||
if (!/- \[x\]/i.test(typeBlock)) {
|
||||
problems.push('**Type of Change** — check at least one box.');
|
||||
descriptionProblems.push('**Type of Change** — check at least one box.');
|
||||
}
|
||||
|
||||
// 4. Duplicate-search checklist item must be checked.
|
||||
if (!/- \[x\] I searched/i.test(body)) {
|
||||
problems.push('**Checklist** — check the duplicate-search box to confirm you searched existing issues and PRs.');
|
||||
descriptionProblems.push('**Checklist** — check the duplicate-search box to confirm you searched existing issues and PRs.');
|
||||
}
|
||||
|
||||
// 5. How to Test must contain enough real detail for a reviewer to act on.
|
||||
@@ -53,7 +53,83 @@ module.exports = async ({ github, context, core }) => {
|
||||
// code block — so we only require non-trivial content, not a specific shape.
|
||||
const howTo = section('How to Test');
|
||||
if (howTo.length < 30) {
|
||||
problems.push('**How to Test** — explain how a reviewer can verify this change. Numbered steps, the commands you ran, or a short code block all work — give a sentence or two of real detail (not just "tested locally").');
|
||||
descriptionProblems.push('**How to Test** — explain how a reviewer can verify this change. Numbered steps, the commands you ran, or a short code block all work — give a sentence or two of real detail (not just "tested locally").');
|
||||
}
|
||||
|
||||
// Classify paths from GitHub's API. This workflow runs in the privileged base
|
||||
// context, so it must never check out or execute code from the PR branch.
|
||||
const changedFiles = await github.paginate(github.rest.pulls.listFiles, {
|
||||
owner, repo, pull_number: prNum, per_page: 100,
|
||||
});
|
||||
const changedPaths = changedFiles.map(file => file.filename);
|
||||
|
||||
function isUiSensitivePath(filename) {
|
||||
const path = filename.toLowerCase();
|
||||
return path.startsWith('static/')
|
||||
|| path.startsWith('templates/')
|
||||
|| /\.(?:html?|css|svg)$/.test(path);
|
||||
}
|
||||
|
||||
function isDocsOnlyPath(filename) {
|
||||
const path = filename.toLowerCase();
|
||||
return /\.(?:md|mdx|rst|adoc|txt)$/.test(path)
|
||||
|| (path.startsWith('docs/') && !isUiSensitivePath(path));
|
||||
}
|
||||
|
||||
function isRuntimeSensitivePath(filename) {
|
||||
const path = filename.toLowerCase();
|
||||
if (isUiSensitivePath(path)) return false;
|
||||
if (path.startsWith('tests/') || path.startsWith('.github/')) return false;
|
||||
return /^(?:app\.py|routes\/|services\/|src\/|core\/|mcp_servers\/|scripts\/|docker\/)/.test(path)
|
||||
|| /^(?:dockerfile|docker-compose.*\.ya?ml|requirements(?:-optional)?\.txt|pyproject\.toml|setup\.py)$/.test(path)
|
||||
|| /\.(?:py|sh|ps1|bat)$/.test(path);
|
||||
}
|
||||
|
||||
let classification = 'tooling';
|
||||
if (changedPaths.some(isUiSensitivePath)) {
|
||||
classification = 'UI-sensitive';
|
||||
} else if (changedPaths.some(isRuntimeSensitivePath)) {
|
||||
classification = 'backend/runtime';
|
||||
} else if (changedPaths.length > 0 && changedPaths.every(isDocsOnlyPath)) {
|
||||
classification = 'docs-only';
|
||||
}
|
||||
|
||||
const appRan = /- \[x\]\s+I actually ran the app\b/i.test(body);
|
||||
const appNotRun = /- \[x\]\s+I did not run the app\/runtime validation\b/i.test(body);
|
||||
// Anchor on the wording, not the template's emphasis: a ticked box the author
|
||||
// retyped without the surrounding ** renders identically on the PR page, so
|
||||
// treating it as unchecked is invisible from their side. Matches the two
|
||||
// attestations above, which already ignore formatting.
|
||||
const screenshotChecked = /- \[x\]\s+[*_]{0,2}Screenshot or short clip[*_]{0,2}/i.test(body);
|
||||
const screenshotSection = section('Screenshots / clips');
|
||||
const hasVisualEvidence = /!\[[^\]]*\]\([^)]+\)|<(?:img|video|source)\b[^>]*(?:src|href)=|https?:\/\/[^\s)]+/i.test(screenshotSection);
|
||||
const evidenceGaps = [];
|
||||
let needsRuntimeValidation = false;
|
||||
let needsVisualEvidence = false;
|
||||
|
||||
if (classification === 'backend/runtime' || classification === 'UI-sensitive') {
|
||||
if (appRan && appNotRun) {
|
||||
needsRuntimeValidation = true;
|
||||
evidenceGaps.push('The app-run and explicit not-run boxes are both checked. Select the one state that is true.');
|
||||
} else if (!appRan) {
|
||||
needsRuntimeValidation = true;
|
||||
if (appNotRun) {
|
||||
evidenceGaps.push('The author explicitly reports that app/runtime validation was not performed.');
|
||||
} else {
|
||||
evidenceGaps.push('App/runtime validation is not author-attested. Check the run box only after running it, or check the explicit not-run box and describe the gap.');
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (classification === 'UI-sensitive') {
|
||||
if (!screenshotChecked) {
|
||||
needsVisualEvidence = true;
|
||||
evidenceGaps.push('The screenshot/clip checkbox is not checked for this UI-sensitive change.');
|
||||
}
|
||||
if (!hasVisualEvidence) {
|
||||
needsVisualEvidence = true;
|
||||
evidenceGaps.push('The Screenshots / clips section does not contain an actual attachment or link.');
|
||||
}
|
||||
}
|
||||
|
||||
// ── Comment ──────────────────────────────────────────────────────────────
|
||||
@@ -62,22 +138,43 @@ module.exports = async ({ github, context, core }) => {
|
||||
});
|
||||
const existing = comments.find(c => (c.body ?? '').includes(MARKER));
|
||||
|
||||
if (problems.length === 0) {
|
||||
if (descriptionProblems.length === 0 && evidenceGaps.length === 0) {
|
||||
if (existing) {
|
||||
await github.rest.issues.deleteComment({ owner, repo, comment_id: existing.id });
|
||||
}
|
||||
} else {
|
||||
const commentBody = [
|
||||
MARKER,
|
||||
'⚠️ **PR description — action needed**',
|
||||
'',
|
||||
'The following required sections are missing or incomplete. Please update the PR description to address them:',
|
||||
'',
|
||||
problems.map(p => `- ${p}`).join('\n'),
|
||||
const commentLines = [MARKER];
|
||||
if (descriptionProblems.length > 0) {
|
||||
commentLines.push(
|
||||
'⚠️ **PR description — action needed**',
|
||||
'',
|
||||
'The following required sections are missing or incomplete. Please update the PR description to address them:',
|
||||
'',
|
||||
descriptionProblems.map(problem => `- ${problem}`).join('\n'),
|
||||
);
|
||||
} else {
|
||||
commentLines.push(
|
||||
'⚠️ **PR description is complete; validation evidence is still outstanding**',
|
||||
'',
|
||||
`Changed-file classification: **${classification}**.`,
|
||||
);
|
||||
}
|
||||
if (evidenceGaps.length > 0) {
|
||||
commentLines.push(
|
||||
'',
|
||||
'**Author-reported runtime / visual state**',
|
||||
'',
|
||||
evidenceGaps.map(gap => `- ${gap}`).join('\n'),
|
||||
'',
|
||||
'Checkboxes are author attestations. GitHub Actions results remain the execution evidence for CI; this check does not prove that a local command ran.',
|
||||
);
|
||||
}
|
||||
commentLines.push(
|
||||
'',
|
||||
'---',
|
||||
'_This comment is deleted automatically once all sections are complete._',
|
||||
].join('\n');
|
||||
'_This comment updates automatically when the description or changed files change._',
|
||||
);
|
||||
const commentBody = commentLines.join('\n');
|
||||
|
||||
if (existing) {
|
||||
await github.rest.issues.updateComment({ owner, repo, comment_id: existing.id, body: commentBody });
|
||||
@@ -97,34 +194,47 @@ module.exports = async ({ github, context, core }) => {
|
||||
return true;
|
||||
} catch (e) {
|
||||
if (e.status === 404) return false;
|
||||
if (e.status === 403) {
|
||||
core.warning(`Could not inspect label "${name}" — token lacks label read access; skipping.`);
|
||||
return false;
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
async function swapLabel(num, add, remove) {
|
||||
if (await labelExists(add)) {
|
||||
async function setLabel(name, wanted) {
|
||||
if (wanted && await labelExists(name)) {
|
||||
try {
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: num, labels: [add] });
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: prNum, labels: [name] });
|
||||
} catch (e) {
|
||||
// Fail soft on a token that can't write labels so a label permission
|
||||
// problem never masks the actual description verdict.
|
||||
if (e.status !== 403) throw e;
|
||||
core.warning(`Could not add "${add}" — token lacks label write here; skipping.`);
|
||||
if (e.status !== 403 && e.status !== 404) throw e;
|
||||
core.warning(`Could not add "${name}" — label is unavailable or the token lacks label write access; skipping.`);
|
||||
}
|
||||
} else if (wanted) {
|
||||
core.warning(`Label "${name}" does not exist in the repo — skipping. Create it once to enable labelling.`);
|
||||
} else {
|
||||
core.warning(`Label "${add}" does not exist in the repo — skipping. Create it once to enable labelling.`);
|
||||
}
|
||||
try {
|
||||
await github.rest.issues.removeLabel({ owner, repo, issue_number: num, name: remove });
|
||||
} catch (e) {
|
||||
if (e.status !== 404 && e.status !== 410 && e.status !== 403) throw e;
|
||||
try {
|
||||
await github.rest.issues.removeLabel({ owner, repo, issue_number: prNum, name });
|
||||
} catch (e) {
|
||||
if (e.status !== 404 && e.status !== 410 && e.status !== 403) throw e;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (problems.length === 0) {
|
||||
await swapLabel(prNum, 'ready for review', 'needs work');
|
||||
} else {
|
||||
await swapLabel(prNum, 'needs work', 'ready for review');
|
||||
core.setFailed(`PR description has ${problems.length} issue(s) — see bot comment for details.`);
|
||||
const descriptionComplete = descriptionProblems.length === 0;
|
||||
const evidenceComplete = evidenceGaps.length === 0;
|
||||
const isDraft = Boolean(context.payload.pull_request.draft);
|
||||
await setLabel(
|
||||
'ready for review',
|
||||
descriptionComplete && evidenceComplete && !isDraft,
|
||||
);
|
||||
await setLabel('needs work', !descriptionComplete);
|
||||
await setLabel('needs runtime validation', needsRuntimeValidation);
|
||||
await setLabel('needs visual evidence', needsVisualEvidence);
|
||||
|
||||
if (!descriptionComplete) {
|
||||
core.setFailed(`PR description has ${descriptionProblems.length} issue(s) — see bot comment for details.`);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Report focused pytest guidance for changed paths under tests/."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
from collections.abc import Iterable
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
|
||||
def parse_paths(raw_paths: bytes) -> list[str]:
|
||||
"""Decode the NUL-delimited output of ``git diff --name-only -z``."""
|
||||
return [os.fsdecode(path) for path in raw_paths.split(b"\0") if path]
|
||||
|
||||
|
||||
def changed_paths_from_merge_base(base_sha: str, head_sha: str) -> list[str]:
|
||||
"""Return changed ``tests/`` paths using GitHub PR three-dot semantics.
|
||||
|
||||
GitHub PR changed files are based on the merge base and the PR head, not a
|
||||
direct endpoint diff between the current base branch tip and the PR head.
|
||||
Using the direct endpoint diff can include files changed only on the base
|
||||
branch when the PR branch is stale.
|
||||
"""
|
||||
merge_base = subprocess.check_output(
|
||||
["git", "merge-base", base_sha, head_sha],
|
||||
stderr=subprocess.DEVNULL,
|
||||
).strip()
|
||||
raw_paths = subprocess.check_output(
|
||||
[
|
||||
"git",
|
||||
"diff",
|
||||
"--name-only",
|
||||
"--diff-filter=ACMRT",
|
||||
"-z",
|
||||
os.fsdecode(merge_base),
|
||||
head_sha,
|
||||
"--",
|
||||
"tests/",
|
||||
],
|
||||
)
|
||||
return parse_paths(raw_paths)
|
||||
|
||||
|
||||
def select_test_paths(paths: Iterable[str]) -> list[str]:
|
||||
"""Return unique, repository-relative paths contained by tests/."""
|
||||
selected: set[str] = set()
|
||||
for raw_path in paths:
|
||||
path = PurePosixPath(raw_path)
|
||||
if path.is_absolute() or ".." in path.parts:
|
||||
continue
|
||||
parts = tuple(part for part in path.parts if part != ".")
|
||||
if len(parts) >= 2 and parts[0] == "tests":
|
||||
selected.add(PurePosixPath(*parts).as_posix())
|
||||
return sorted(selected)
|
||||
|
||||
|
||||
def is_pytest_file(path: str) -> bool:
|
||||
"""Return whether a changed path follows this repository's pytest naming."""
|
||||
name = PurePosixPath(path).name
|
||||
return name.endswith(".py") and (
|
||||
name.startswith("test_") or name.endswith("_test.py")
|
||||
)
|
||||
|
||||
|
||||
def pytest_command(paths: Iterable[str]) -> str:
|
||||
"""Build a copyable pytest command for changed runnable test files."""
|
||||
command = ["python3", "-m", "pytest", "-q", *paths]
|
||||
return shlex.join(command)
|
||||
|
||||
|
||||
def format_report(paths: Iterable[str]) -> str:
|
||||
"""Format focused guidance for CI logs and the workflow summary."""
|
||||
changed_paths = select_test_paths(paths)
|
||||
runnable_paths = [path for path in changed_paths if is_pytest_file(path)]
|
||||
lines = ["## Focused test guidance (report-only)", ""]
|
||||
if not changed_paths:
|
||||
lines.append("No changed paths under `tests/`.")
|
||||
else:
|
||||
lines.extend(["Changed paths under `tests/`:", ""])
|
||||
lines.extend(f"- `{path}`" for path in changed_paths)
|
||||
lines.extend(["", "Suggested focused validation:", ""])
|
||||
if runnable_paths:
|
||||
lines.append(f"```sh\n{pytest_command(runnable_paths)}\n```")
|
||||
else:
|
||||
lines.append("No directly runnable pytest files changed.")
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"This guidance does not infer tests from source changes. "
|
||||
"Existing blocking CI remains the source of truth.",
|
||||
]
|
||||
)
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _parse_args(argv: list[str]) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Report focused pytest guidance for changed tests/ paths.",
|
||||
)
|
||||
parser.add_argument("--base-sha", help="Pull request base commit SHA.")
|
||||
parser.add_argument("--head-sha", help="Pull request head commit SHA.")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _parse_args(sys.argv[1:] if argv is None else argv)
|
||||
if bool(args.base_sha) != bool(args.head_sha):
|
||||
raise SystemExit("--base-sha and --head-sha must be provided together")
|
||||
|
||||
if args.base_sha and args.head_sha:
|
||||
paths = changed_paths_from_merge_base(args.base_sha, args.head_sha)
|
||||
else:
|
||||
paths = parse_paths(sys.stdin.buffer.read())
|
||||
|
||||
print(format_report(paths))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+66
-14
@@ -2,7 +2,7 @@ name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
branches: [main, dev]
|
||||
pull_request:
|
||||
|
||||
# Least privilege: none of the jobs write to the repo.
|
||||
@@ -15,14 +15,68 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
focused-test-guidance:
|
||||
name: Focused test guidance (report-only)
|
||||
if: github.event_name == 'pull_request'
|
||||
runs-on: ubuntu-latest
|
||||
continue-on-error: true
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
- name: Report changed test paths
|
||||
env:
|
||||
BASE_SHA: ${{ github.event.pull_request.base.sha }}
|
||||
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
run: |
|
||||
report_file="$RUNNER_TEMP/focused-test-guidance.md"
|
||||
publish_report() {
|
||||
cat "$report_file"
|
||||
if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
|
||||
cat "$report_file" >> "$GITHUB_STEP_SUMMARY" || true
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
report_unavailable() {
|
||||
{
|
||||
printf '%s\n\n' '## Focused test guidance unavailable (report-only)'
|
||||
printf '%s\n\n' "$1"
|
||||
printf '%s\n' 'Existing blocking CI remains the source of truth.'
|
||||
} > "$report_file"
|
||||
publish_report
|
||||
exit 0
|
||||
}
|
||||
|
||||
if [ -z "$BASE_SHA" ] || [ -z "$HEAD_SHA" ]; then
|
||||
report_unavailable "Pull request base/head metadata is missing."
|
||||
fi
|
||||
|
||||
if ! git cat-file -e "${BASE_SHA}^{commit}" 2>/dev/null; then
|
||||
report_unavailable "The pull request base commit is unavailable locally."
|
||||
fi
|
||||
|
||||
if ! git cat-file -e "${HEAD_SHA}^{commit}" 2>/dev/null; then
|
||||
report_unavailable "The pull request head commit is unavailable locally."
|
||||
fi
|
||||
|
||||
if ! python3 .github/scripts/focused_test_guidance.py \
|
||||
--base-sha "$BASE_SHA" \
|
||||
--head-sha "$HEAD_SHA" > "$report_file"; then
|
||||
report_unavailable "The focused test guidance helper could not produce a report."
|
||||
fi
|
||||
|
||||
publish_report
|
||||
|
||||
python-syntax:
|
||||
name: Python syntax (compileall)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: "3.11"
|
||||
# Byte-compile sources — catches syntax errors without installing deps.
|
||||
@@ -32,10 +86,10 @@ jobs:
|
||||
name: JS syntax (node --check)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: "20"
|
||||
# Syntax-check our own JS (skip vendored libs in static/lib).
|
||||
@@ -49,17 +103,14 @@ jobs:
|
||||
python-tests:
|
||||
name: Python tests (pytest)
|
||||
runs-on: ubuntu-latest
|
||||
# Informational for now: the suite has known flaky / environment-dependent
|
||||
# failures (test isolation + embedding-model assertions). Tracked under the
|
||||
# ROADMAP "fresh install smoke tests" item; make this required once green.
|
||||
continue-on-error: true
|
||||
# Make Python test validation authoritative for the configured scope.
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
|
||||
# Detect whether this PR only touches documentation files.
|
||||
# Detect whether this PR only touches repository prose outside the Pages site.
|
||||
# If so, skip the expensive pytest run while still reporting a passing check.
|
||||
- name: Check for docs-only changes
|
||||
id: docs-check
|
||||
@@ -71,9 +122,10 @@ jobs:
|
||||
BASE="${{ github.event.before }}"
|
||||
HEAD="${{ github.sha }}"
|
||||
fi
|
||||
# List all changed files; if every file matches docs/markdown patterns, skip pytest.
|
||||
# Keep website/ and assets/branding/ out of this bypass: pytest owns
|
||||
# regression guards for their published-file and orphan-asset contracts.
|
||||
changed=$(git diff --name-only "$BASE" "$HEAD" 2>/dev/null || git diff --name-only HEAD~1 HEAD)
|
||||
non_docs=$(echo "$changed" | grep -Ev '^(docs/|.*\.md$|\.github/[^/]+\.md$)' || true)
|
||||
non_docs=$(echo "$changed" | grep -Ev '^(docs/|[^/]+\.md$|\.github/[^/]+\.md$)' || true)
|
||||
if [ -z "$non_docs" ]; then
|
||||
echo "docs_only=true" >> "$GITHUB_OUTPUT"
|
||||
echo "Docs-only change detected — skipping pytest."
|
||||
@@ -81,7 +133,7 @@ jobs:
|
||||
echo "docs_only=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
if: steps.docs-check.outputs.docs_only != 'true'
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
name: CodeQL
|
||||
|
||||
# Advanced setup so CodeQL also runs on pull requests (including from forks),
|
||||
# surfacing findings before merge instead of only after a change lands on dev.
|
||||
on:
|
||||
push:
|
||||
branches: [dev, main]
|
||||
pull_request:
|
||||
branches: [dev]
|
||||
schedule:
|
||||
- cron: "17 3 * * 1"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
analyze:
|
||||
name: Analyze (${{ matrix.language }})
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
security-events: write
|
||||
actions: read
|
||||
contents: read
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
language: [actions, javascript-typescript, python]
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
|
||||
with:
|
||||
languages: ${{ matrix.language }}
|
||||
build-mode: none
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
|
||||
with:
|
||||
category: "/language:${{ matrix.language }}"
|
||||
@@ -37,12 +37,12 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Lint Dockerfile
|
||||
uses: hadolint/hadolint-action@2332a7b74a6de0dda2e2221d575162eba76ba5e5 # v3.3.0
|
||||
uses: hadolint/hadolint-action@2a66e89f53d0771bb131a7fa31f3136336094aa6 # v3.4.0
|
||||
with:
|
||||
dockerfile: Dockerfile
|
||||
# DL3008: pinning apt package versions is impractical on a -slim base
|
||||
|
||||
@@ -23,12 +23,16 @@ on:
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs/**'
|
||||
- 'website/**'
|
||||
- 'assets/branding/**'
|
||||
- '.github/ISSUE_TEMPLATE/**'
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs/**'
|
||||
- 'website/**'
|
||||
- 'assets/branding/**'
|
||||
- '.github/ISSUE_TEMPLATE/**'
|
||||
workflow_dispatch:
|
||||
|
||||
@@ -52,17 +56,17 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Buildx
|
||||
uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4.1.0
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
|
||||
|
||||
# Build without pushing so a broken Dockerfile is caught here, and the
|
||||
# exact image we ship is what gets scanned.
|
||||
- name: Build image
|
||||
uses: docker/build-push-action@f9f3042f7e2789586610d6e8b85c8f03e5195baf # v7.2.0
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
|
||||
with:
|
||||
context: .
|
||||
push: false
|
||||
@@ -93,15 +97,15 @@ jobs:
|
||||
security-events: write # upload SARIF to the Security tab
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Buildx
|
||||
uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4.1.0
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
|
||||
|
||||
- name: Build image
|
||||
uses: docker/build-push-action@f9f3042f7e2789586610d6e8b85c8f03e5195baf # v7.2.0
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
|
||||
with:
|
||||
context: .
|
||||
push: false
|
||||
@@ -119,7 +123,7 @@ jobs:
|
||||
TRIVY_DB_REPOSITORY: ghcr.io/aquasecurity/trivy-db:2
|
||||
|
||||
- name: Upload Trivy results
|
||||
uses: github/codeql-action/upload-sarif@8aad20d150bbac5944a9f9d289da16a4b0d87c1e # v4.36.2
|
||||
uses: github/codeql-action/upload-sarif@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-image
|
||||
|
||||
@@ -36,7 +36,7 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
@@ -55,12 +55,12 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
name: Deploy GitHub Pages
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'website/**'
|
||||
- '.github/workflows/deploy-pages.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions: {}
|
||||
|
||||
concurrency:
|
||||
group: pages
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Package static site
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
pages: read
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0
|
||||
- uses: actions/jekyll-build-pages@44a6e6beabd48582f863aeeb6cb2151cc1716697 # v1.0.13
|
||||
with:
|
||||
source: website
|
||||
destination: _site
|
||||
- uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
with:
|
||||
path: _site
|
||||
|
||||
deploy:
|
||||
name: Deploy static site
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
pages: write
|
||||
id-token: write
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deployment.outputs.page_url }}
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
@@ -14,6 +14,8 @@ on:
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs/**'
|
||||
- 'website/**'
|
||||
- 'assets/branding/**'
|
||||
- '.github/ISSUE_TEMPLATE/**'
|
||||
|
||||
concurrency:
|
||||
@@ -45,20 +47,20 @@ jobs:
|
||||
arch: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up Buildx
|
||||
uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4.1.0
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4.2.0
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Build and push by digest
|
||||
id: build
|
||||
uses: docker/build-push-action@f9f3042f7e2789586610d6e8b85c8f03e5195baf # v7.2.0
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
|
||||
with:
|
||||
context: .
|
||||
platforms: ${{ matrix.platform }}
|
||||
@@ -86,7 +88,7 @@ jobs:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Read APP_VERSION + short sha
|
||||
@@ -103,16 +105,16 @@ jobs:
|
||||
pattern: digest-*
|
||||
merge-multiple: true
|
||||
- name: Set up Buildx
|
||||
uses: docker/setup-buildx-action@d7f5e7f509e45cec5c76c4d5afdd7de93d0b3df5 # v4.1.0
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4.2.0
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Compute tags
|
||||
id: meta
|
||||
uses: docker/metadata-action@80c7e94dd9b9319bd5eb7a0e0fe9291e23a2a2e9 # v6.1.0
|
||||
uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6.2.0
|
||||
with:
|
||||
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
|
||||
tags: |
|
||||
|
||||
@@ -2,7 +2,7 @@ name: ci / issue description check
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened, edited, reopened]
|
||||
types: [opened, edited, reopened, closed]
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
@@ -14,7 +14,7 @@ jobs:
|
||||
# Skip bots (Dependabot, release-drafter, etc.)
|
||||
if: ${{ github.event.issue.user.type != 'Bot' }}
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
sparse-checkout: .github/scripts
|
||||
persist-credentials: false
|
||||
|
||||
@@ -5,7 +5,11 @@ on:
|
||||
# works on fork PRs. Safe here: the checkout pins to the base branch (no fork
|
||||
# code runs) and the scripts only read context.payload and call the GitHub API.
|
||||
pull_request_target: # zizmor: ignore[dangerous-triggers]
|
||||
types: [opened, edited, synchronize, reopened, ready_for_review]
|
||||
types: [opened, edited, synchronize, reopened, ready_for_review, converted_to_draft]
|
||||
|
||||
concurrency:
|
||||
group: pr-description-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Default-deny at the workflow level; each job opts into only the scopes it needs.
|
||||
# Note: modifying a PR's labels/comments needs pull-requests:write even though the
|
||||
@@ -23,7 +27,7 @@ jobs:
|
||||
# Skip bots: they open PRs programmatically and have their own process.
|
||||
if: github.event.pull_request.user.type != 'Bot'
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
ref: ${{ github.base_ref }}
|
||||
sparse-checkout: .github/scripts
|
||||
@@ -59,12 +63,14 @@ jobs:
|
||||
|
||||
check-mergeable:
|
||||
name: Flag unmergeable PRs
|
||||
needs: check-description
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
pull-requests: write
|
||||
issues: write
|
||||
# Skip bots: they open PRs programmatically and have their own process.
|
||||
if: github.event.pull_request.user.type != 'Bot'
|
||||
# Run after description validation failures, but never from an obsolete
|
||||
# workflow run canceled by a newer PR event.
|
||||
if: ${{ !cancelled() && github.event.pull_request.user.type != 'Bot' }}
|
||||
steps:
|
||||
- uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
|
||||
with:
|
||||
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
# Full history so a secret committed in an earlier commit (and later
|
||||
# deleted) is still caught -- deletion does not remove it from Git.
|
||||
|
||||
@@ -36,7 +36,7 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
@@ -61,12 +61,12 @@ jobs:
|
||||
contents: read
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
|
||||
+18
@@ -85,6 +85,24 @@ output.txt.txt
|
||||
!docs/**/*.gif
|
||||
!docs/**/*.webp
|
||||
|
||||
# …except shipped website and branding media.
|
||||
!website/**/*.jpg
|
||||
!website/**/*.jpeg
|
||||
!website/**/*.png
|
||||
!website/**/*.gif
|
||||
!website/**/*.bmp
|
||||
!website/**/*.webp
|
||||
!website/**/*.tiff
|
||||
!website/**/*.pdf
|
||||
!assets/branding/**/*.jpg
|
||||
!assets/branding/**/*.jpeg
|
||||
!assets/branding/**/*.png
|
||||
!assets/branding/**/*.gif
|
||||
!assets/branding/**/*.bmp
|
||||
!assets/branding/**/*.webp
|
||||
!assets/branding/**/*.tiff
|
||||
!assets/branding/**/*.pdf
|
||||
|
||||
# Reports and temp files
|
||||
reports/
|
||||
tasks/
|
||||
|
||||
+10
-2
@@ -65,6 +65,16 @@ Vendored in `static/lib/` and served directly:
|
||||
| [jsPDF](https://github.com/parallax/jsPDF) (bundled in html2pdf) | PDF generation | MIT |
|
||||
| [html2canvas](https://github.com/niklasvh/html2canvas) (bundled in html2pdf) | DOM → canvas rasterization | MIT |
|
||||
| [node-qrcode](https://github.com/soldair/node-qrcode) (`qrcode.min.js`) | QR-code rendering (2FA setup) | MIT |
|
||||
| [KaTeX](https://github.com/KaTeX/KaTeX) v0.16.22 (`katex/katex.min.{js,css}` + `katex/fonts/*.woff2`) | Math typesetting | MIT ([`licenses/KaTeX-MIT-LICENSE.txt`](licenses/KaTeX-MIT-LICENSE.txt)) |
|
||||
| [Mermaid](https://github.com/mermaid-js/mermaid) v11.16.1 (`mermaid.min.js`) | Diagrams from text | MIT ([`licenses/Mermaid-MIT-LICENSE.txt`](licenses/Mermaid-MIT-LICENSE.txt)) |
|
||||
|
||||
KaTeX and Mermaid are loaded on first use by `static/js/markdown.js` rather than
|
||||
from `index.html`, so a session that renders no math and no diagram never fetches
|
||||
either. Only the `.woff2` KaTeX fonts are shipped, matching `static/fonts/`; the
|
||||
`.woff` and `.ttf` variants its stylesheet also lists are never requested by a
|
||||
browser that supports `woff2`. The bundles are the published npm artifacts,
|
||||
unmodified — `.gitattributes` turns the whitespace check off for `static/lib/`
|
||||
so they can stay byte-identical to upstream.
|
||||
|
||||
## Front-end libraries loaded at runtime (CDN)
|
||||
|
||||
@@ -72,8 +82,6 @@ Referenced from `cdn.jsdelivr.net` / `cdnjs.cloudflare.com` at runtime — not v
|
||||
|
||||
| Library | Purpose | License |
|
||||
|---|---|---|
|
||||
| [KaTeX](https://github.com/KaTeX/KaTeX) 0.16.22 | Math typesetting | MIT |
|
||||
| [Mermaid](https://github.com/mermaid-js/mermaid) 11 | Diagrams from text | MIT |
|
||||
| [Pyodide](https://github.com/pyodide/pyodide) 0.27.5 | In-browser Python runtime | MPL-2.0 |
|
||||
| [PDFObject](https://github.com/pipwerks/PDFObject) 2.1.1 | Inline PDF embedding | MIT |
|
||||
|
||||
|
||||
+1
-1
@@ -25,7 +25,7 @@ End-users cloning the repo will land on `dev` by default. To run the curated/sta
|
||||
Docker is the recommended path for normal testing:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/pewdiepie-archdaemon/odysseus.git
|
||||
git clone https://github.com/odysseus-dev/odysseus.git
|
||||
cd odysseus
|
||||
cp .env.example .env
|
||||
docker compose up -d --build
|
||||
|
||||
+20
-2
@@ -16,7 +16,12 @@ FROM python:3.14-slim
|
||||
# downloads, and serves from Docker installs.
|
||||
# git/cmake are required when Cookbook builds llama.cpp on first llama.cpp
|
||||
# launch inside Docker.
|
||||
# nodejs/npm provide npx for the optional built-in Browser MCP server.
|
||||
# nodejs/npm provide npx for the built-in Browser MCP server.
|
||||
# chromium provides the actual browser binary used by that MCP server.
|
||||
# fontconfig + Noto CJK provide real fallback glyphs for multilingual pages;
|
||||
# Chromium otherwise renders Chinese/Japanese/Korean labels as empty boxes.
|
||||
# iproute2/iputils-ping/net-tools/dnsutils/nmap give Docker-hosted agents the
|
||||
# basic network inspection toolkit expected by local LAN/debugging tasks.
|
||||
# gosu lets the entrypoint drop privileges cleanly so signals still reach
|
||||
# uvicorn directly (no extra shell layer like `su`/`sudo` would add).
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
@@ -26,8 +31,16 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git \
|
||||
nodejs \
|
||||
npm \
|
||||
chromium \
|
||||
fontconfig \
|
||||
fonts-noto-cjk \
|
||||
tmux \
|
||||
openssh-client \
|
||||
iproute2 \
|
||||
iputils-ping \
|
||||
net-tools \
|
||||
dnsutils \
|
||||
nmap \
|
||||
gosu \
|
||||
libgl1 \
|
||||
libglib2.0-0t64 \
|
||||
@@ -35,6 +48,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libmagic1 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Private browser automation wrapper used by the native `private_browser` tool.
|
||||
# Chromium is installed above, so agent-browser can drive the existing browser
|
||||
# binary without paying `npx` startup/install overhead on each tool call.
|
||||
RUN npm install -g agent-browser@0.35.0 --omit=dev --loglevel=error
|
||||
|
||||
# libgl1/libglib2.0-0t64/libxcb1 are runtime shared libs (libGL.so.1,
|
||||
# libglib-2.0/libgthread, libxcb.so.1) that opencv-python (cv2) loads. The
|
||||
# slim base omits them, so the Cookbook "install realesrgan" path imports cv2
|
||||
@@ -54,7 +72,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
# /var/run/docker.sock mount). The Debian `docker.io` package ships
|
||||
# dockerd but not the client binary on slim, so grab the static client
|
||||
# tarball from download.docker.com instead.
|
||||
ARG DOCKER_CLI_VERSION=27.5.1
|
||||
ARG DOCKER_CLI_VERSION=29.6.2
|
||||
RUN ARCH="$(dpkg --print-architecture)" \
|
||||
&& case "$ARCH" in \
|
||||
amd64) DARCH=x86_64 ;; \
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
0.20.5
|
||||
@@ -1,5 +1,5 @@
|
||||
<p align="center">
|
||||
<img src="docs/odysseus-wordmark.png" alt="Odysseus" width="238">
|
||||
<img src="assets/branding/odysseus-wordmark.png" alt="Odysseus" width="238">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -8,7 +8,7 @@
|
||||
|
||||
<p align="center">
|
||||
<a href="#quick-start">Quick Start</a> ·
|
||||
<a href="docs/setup.md">Setup Guide</a> ·
|
||||
<a href="website/setup.md">Setup Guide</a> ·
|
||||
<a href="CONTRIBUTING.md">Contributing</a> ·
|
||||
<a href="ROADMAP.md">Roadmap</a>
|
||||
</p>
|
||||
@@ -18,17 +18,17 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/odysseus-browser.jpg" alt="Odysseus interface">
|
||||
<img src="assets/branding/odysseus-browser.jpg" alt="Odysseus interface">
|
||||
</p>
|
||||
|
||||
---
|
||||
|
||||
## Quick Start
|
||||
|
||||
> `dev` is the default branch and gets the newest changes first. Use [`main`](https://github.com/pewdiepie-archdaemon/odysseus/tree/main) if you want the more curated branch.
|
||||
> `dev` is the default branch and gets the newest changes first. Use [`main`](https://github.com/odysseus-dev/odysseus/tree/main) if you want the more curated branch.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/pewdiepie-archdaemon/odysseus.git
|
||||
git clone https://github.com/odysseus-dev/odysseus.git
|
||||
cd odysseus
|
||||
cp .env.example .env
|
||||
docker compose up -d --build
|
||||
@@ -36,7 +36,7 @@ docker compose up -d --build
|
||||
|
||||
Open `http://localhost:7000` when the containers are healthy. The first admin password is printed in `docker compose logs odysseus`.
|
||||
|
||||
Native installs, GPU notes, Windows/macOS instructions, HTTPS, and configuration live in the [setup guide](docs/setup.md).
|
||||
Native installs, GPU notes, Windows/macOS instructions, HTTPS, and configuration live in the [setup guide](website/setup.md).
|
||||
|
||||
## Features
|
||||
|
||||
@@ -51,7 +51,7 @@ Native installs, GPU notes, Windows/macOS instructions, HTTPS, and configuration
|
||||
|
||||
## Demo
|
||||
|
||||
A full hover-to-play tour lives on the landing page: [`docs/index.html`](docs/index.html).
|
||||
A full hover-to-play tour lives on the [Odysseus landing page](https://odysseus-dev.github.io/odysseus/). Its source lives under [`website/`](website/).
|
||||
|
||||
## Contributing
|
||||
|
||||
@@ -59,15 +59,20 @@ Help is welcome. The best entry points are fresh-install testing, provider setup
|
||||
|
||||
## Security
|
||||
|
||||
Odysseus is a self-hosted workspace with powerful local tools. Keep auth enabled, keep private data out of Git, and do not expose raw model/service ports publicly. Deployment details are in the [setup guide](docs/setup.md#security-notes).
|
||||
Odysseus is a self-hosted workspace with powerful local tools. Keep auth enabled, keep private data out of Git, and do not expose raw model/service ports publicly.
|
||||
|
||||
- Keep `AUTH_ENABLED=true` for any network-accessible deployment.
|
||||
- Keep `LOCALHOST_BYPASS=false` outside local development.
|
||||
|
||||
Deployment details are in the [setup guide](website/setup.md#security-notes).
|
||||
|
||||
## Star History
|
||||
|
||||
<a href="https://www.star-history.com/?repos=pewdiepie-archdaemon%2Fodysseus&type=date&legend=top-left">
|
||||
<a href="https://star-history.dera.page/#odysseus-dev/odysseus&type=date&legend=top-left">
|
||||
<picture>
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/chart?repos=pewdiepie-archdaemon/odysseus&type=date&theme=dark&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/chart?repos=pewdiepie-archdaemon/odysseus&type=date&legend=top-left" />
|
||||
<img alt="Star History Chart" src="https://api.star-history.com/chart?repos=pewdiepie-archdaemon/odysseus&type=date&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://star-history.dera.page/svg?repos=odysseus-dev/odysseus&type=date&theme=dark&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://star-history.dera.page/svg?repos=odysseus-dev/odysseus&type=date&legend=top-left" />
|
||||
<img alt="Star History Chart" src="https://star-history.dera.page/svg?repos=odysseus-dev/odysseus&type=date&legend=top-left" />
|
||||
</picture>
|
||||
</a>
|
||||
|
||||
|
||||
+8
-1
@@ -12,7 +12,6 @@ the codebase, you are probably right to stay away.
|
||||
and WSL all need coverage.
|
||||
|
||||
- Integration audit: do integrations even work? Confirm what works, what needs setup docs, and what should be removed or hidden.
|
||||
- Self-host troubleshooting cookbook. Document the weird 30-second fixes that otherwise become 30-minute searches: Dovecot cleartext auth for local stacks, ntfy Android Instant Delivery for non-ntfy.sh servers, clipboard limits on plain-HTTP Tailscale URLs, Radicale collection URLs, and similar traps.
|
||||
- Cookbook reliability on other computers. This is probably the area most likely to need work across different machines, GPUs, drivers, shells, and Python environments.
|
||||
- Cookbook SGLang support across platforms. Make sure SGLang setup/serve works
|
||||
predictably on Linux, Windows/WSL, macOS where possible, Docker, and common
|
||||
@@ -33,6 +32,14 @@ the codebase, you are probably right to stay away.
|
||||
before the user request really starts. We need slimmer prompts, better tool
|
||||
selection, smaller default tool sets, and clearer guidance for models with
|
||||
4k/8k/16k context windows.
|
||||
- Local model speculative decoding support. For Odysseus-tuned local models,
|
||||
plan to ship or recommend a small same-tokenizer draft model when the serving
|
||||
backend supports it. Early vLLM testing showed a generic `Qwen3-0.6B` draft
|
||||
beside `Qwen3-8B` can materially reduce wall time, while an unsupported
|
||||
DSpark conversion performed poorly. Treat this as a supported draft-model lane
|
||||
first; keep MTP-specific packaging as future work only when the architecture
|
||||
and runtime support are real. Judge this by time-to-success, tool correctness,
|
||||
grammar, and unchanged target output, not tokens/sec alone.
|
||||
- Skill/tool prompt-injection audit. User-editable skills, notes, documents,
|
||||
fetched pages, and memories should be treated as untrusted data. Keep testing
|
||||
whether models follow malicious instructions from those surfaces.
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ Security fixes are handled on the default branch until formal releases are cut.
|
||||
|
||||
- Keep `AUTH_ENABLED=true` for any network-accessible deployment.
|
||||
- Keep `LOCALHOST_BYPASS=false` outside local development.
|
||||
- Set `SECURE_COOKIES=true` when Odysseus is served through HTTPS by a trusted reverse proxy or private access gateway.
|
||||
- Leave `SECURE_COOKIES` unset unless you need to override it: session cookies are marked `Secure` whenever the request arrives over HTTPS. Set `SECURE_COOKIES=true` to force it on (for a proxy Odysseus cannot see the scheme of), or `SECURE_COOKIES=false` to force it off while you still serve plain HTTP alongside HTTPS.
|
||||
- Use HTTPS when exposing the app beyond localhost.
|
||||
- Put the authenticated Odysseus web/API entrypoint behind a trusted reverse proxy or private access layer such as Cloudflare Access, Tailscale, or a VPN.
|
||||
- Keep ChromaDB, SearXNG, ntfy, Ollama, vLLM, llama.cpp, databases, and raw model/provider APIs internal-only.
|
||||
|
||||
+1
-1
@@ -37,7 +37,7 @@ Non-admin defaults are in `core/auth.py:DEFAULT_PRIVILEGES`. Tool enforcement is
|
||||
|
||||
- **Sessions:** bcrypt passwords, 7-day session tokens stored atomically in `data/sessions.json` via `core/atomic_io.py`.
|
||||
- **2FA:** TOTP with 8 single-use backup codes. Verified after password check, before session issuance.
|
||||
- **Reserved usernames:** `internal-tool`, `api`, `demo`, `system` cannot be registered or renamed into. Defined in `core/auth.py:RESERVED_USERNAMES`.
|
||||
- **Reserved usernames:** request sentinels and the Default/Local storage owner cannot be registered or renamed into. Defined in `core/auth.py:RESERVED_USERNAMES`.
|
||||
- `internal-tool` is security-critical: `core/middleware.py:require_admin` treats any request where `request.state.current_user == "internal-tool"` as the in-process tool loopback and grants admin unconditionally. A real account with that name would silently pass every `require_admin` check.
|
||||
- **Orphan sessions:** `validate_token` re-checks that the user record still exists on every call. A deleted user's cookie is dropped on next request rather than continuing to authenticate.
|
||||
|
||||
|
||||
@@ -3,6 +3,9 @@ import mimetypes
|
||||
import os
|
||||
import sys
|
||||
import asyncio
|
||||
import time
|
||||
import shutil
|
||||
import socket
|
||||
|
||||
# On Windows, asyncio.create_subprocess_exec/shell require the ProactorEventLoop.
|
||||
# When started via `python -m uvicorn` from a terminal, uvicorn sets this
|
||||
@@ -66,7 +69,13 @@ from core.constants import (
|
||||
REQUEST_TIMEOUT, OPENAI_API_KEY, AUTH_FILE,
|
||||
)
|
||||
from core.database import SessionLocal, ApiToken
|
||||
from core.middleware import SecurityHeadersMiddleware, is_cors_preflight
|
||||
from core.middleware import (
|
||||
SecurityHeadersMiddleware,
|
||||
get_application_route_path,
|
||||
is_cors_preflight,
|
||||
path_is_route_or_child,
|
||||
with_asgi_root_path,
|
||||
)
|
||||
from core.auth import AuthManager, normalize_known_username
|
||||
from core.exceptions import (
|
||||
SessionNotFoundError, InvalidFileUploadError,
|
||||
@@ -77,6 +86,7 @@ import bcrypt as _bcrypt
|
||||
|
||||
from src.app_helpers import abs_join, serve_html_with_nonce
|
||||
from src.generated_images import GENERATED_IMAGE_HEADERS, resolve_generated_image_path
|
||||
from src.owner_identity import auth_disabled
|
||||
from starlette.responses import RedirectResponse
|
||||
|
||||
# ========= LOGGING =========
|
||||
@@ -152,7 +162,8 @@ app.add_middleware(
|
||||
# model-probe — all served with media_type="text/event-stream") are never
|
||||
# compressed or buffered; only complete bodies over minimum_size are. The
|
||||
# security-header middleware composes cleanly on top.
|
||||
app.add_middleware(GZipMiddleware, minimum_size=1024, compresslevel=6)
|
||||
if os.getenv("RESPONSE_COMPRESSION_ENABLED", "true").strip().lower() not in {"0", "false", "no", "off"}:
|
||||
app.add_middleware(GZipMiddleware, minimum_size=1024, compresslevel=6)
|
||||
|
||||
# ========= SECURITY HEADERS MIDDLEWARE =========
|
||||
app.add_middleware(SecurityHeadersMiddleware)
|
||||
@@ -197,14 +208,57 @@ class _RequestTimeoutMiddleware(_BaseHTTPMiddleware):
|
||||
)
|
||||
|
||||
|
||||
class _InteractiveActivityMiddleware(_BaseHTTPMiddleware):
|
||||
async def dispatch(self, request, call_next):
|
||||
from src.interactive_gate import should_track_interactive_request, track_interactive_request
|
||||
|
||||
path = request.url.path or ""
|
||||
if not should_track_interactive_request(path, request.method):
|
||||
return await call_next(request)
|
||||
async def _stop_background():
|
||||
try:
|
||||
await task_scheduler.stop_background_tasks_for_foreground(reason=f"foreground request {request.method} {path}")
|
||||
except Exception:
|
||||
logging.getLogger("app.foreground_gate").debug("foreground task stop failed", exc_info=True)
|
||||
asyncio.create_task(_stop_background())
|
||||
async with track_interactive_request(path, request.method):
|
||||
return await call_next(request)
|
||||
|
||||
|
||||
class _SlowRequestLogMiddleware(_BaseHTTPMiddleware):
|
||||
async def dispatch(self, request, call_next):
|
||||
start = time.perf_counter()
|
||||
status = 500
|
||||
try:
|
||||
response = await call_next(request)
|
||||
status = getattr(response, "status_code", 0) or 0
|
||||
return response
|
||||
finally:
|
||||
elapsed = time.perf_counter() - start
|
||||
try:
|
||||
threshold = float(os.getenv("ODYSSEUS_SLOW_REQUEST_LOG_SECONDS", "0.75") or "0.75")
|
||||
except Exception:
|
||||
threshold = 0.75
|
||||
if elapsed >= threshold:
|
||||
logging.getLogger("app.slow_request").warning(
|
||||
"slow_request method=%s path=%s status=%s elapsed=%.3fs",
|
||||
request.method,
|
||||
request.url.path,
|
||||
status,
|
||||
elapsed,
|
||||
)
|
||||
|
||||
|
||||
app.add_middleware(_RequestTimeoutMiddleware)
|
||||
app.add_middleware(_InteractiveActivityMiddleware)
|
||||
app.add_middleware(_SlowRequestLogMiddleware)
|
||||
|
||||
# ========= AUTH =========
|
||||
from routes.auth_routes import setup_auth_routes, SESSION_COOKIE
|
||||
|
||||
auth_manager = AuthManager()
|
||||
app.state.auth_manager = auth_manager
|
||||
AUTH_ENABLED = os.getenv("AUTH_ENABLED", "true").lower() != "false"
|
||||
AUTH_ENABLED = not auth_disabled()
|
||||
LOCALHOST_BYPASS = os.getenv("LOCALHOST_BYPASS", "false").lower() == "true"
|
||||
if LOCALHOST_BYPASS:
|
||||
logger.warning("LOCALHOST_BYPASS is enabled, loopback requests bypass authentication. Do not expose this instance to a network.")
|
||||
@@ -240,7 +294,7 @@ if AUTH_ENABLED:
|
||||
def _is_auth_exempt(path: str) -> bool:
|
||||
if path in AUTH_EXEMPT_EXACT:
|
||||
return True
|
||||
if any(path.startswith(p) for p in AUTH_EXEMPT_PREFIXES):
|
||||
if any(path_is_route_or_child(path, p) for p in AUTH_EXEMPT_PREFIXES):
|
||||
return True
|
||||
return any(p.match(path) for p in AUTH_EXEMPT_PATTERNS)
|
||||
|
||||
@@ -311,7 +365,7 @@ if AUTH_ENABLED:
|
||||
|
||||
class AuthMiddleware(BaseHTTPMiddleware):
|
||||
async def dispatch(self, request: Request, call_next):
|
||||
path = request.url.path
|
||||
path = get_application_route_path(request.scope)
|
||||
# A genuine CORS preflight (OPTIONS + Access-Control-Request-Method)
|
||||
# carries no credentials by design and must reach CORSMiddleware to be
|
||||
# answered. AuthMiddleware is the outermost middleware, so gating the
|
||||
@@ -355,7 +409,10 @@ if AUTH_ENABLED:
|
||||
if not auth_manager.is_configured:
|
||||
# No users yet — redirect to login for first-time setup
|
||||
if not path.startswith("/api/"):
|
||||
return RedirectResponse(url="/login", status_code=302)
|
||||
return RedirectResponse(
|
||||
url=with_asgi_root_path(request.scope, "/login"),
|
||||
status_code=302,
|
||||
)
|
||||
return JSONResponse(status_code=401, content={"error": "Setup required"})
|
||||
|
||||
# --- Bearer token auth (API tokens for external integrations) ---
|
||||
@@ -417,7 +474,10 @@ if AUTH_ENABLED:
|
||||
if not auth_manager.validate_token(token):
|
||||
if path.startswith("/api/"):
|
||||
return JSONResponse(status_code=401, content={"error": "Not authenticated"})
|
||||
return RedirectResponse(url="/login", status_code=302)
|
||||
return RedirectResponse(
|
||||
url=with_asgi_root_path(request.scope, "/login"),
|
||||
status_code=302,
|
||||
)
|
||||
|
||||
# Attach current username to request state for downstream routes
|
||||
request.state.current_user = auth_manager.get_username_for_token(token)
|
||||
@@ -583,6 +643,31 @@ webhook_manager = WebhookManager(api_key_manager=api_key_manager)
|
||||
auth_router = setup_auth_routes(auth_manager)
|
||||
app.include_router(auth_router)
|
||||
|
||||
|
||||
@app.post("/api/activity/heartbeat")
|
||||
async def activity_heartbeat():
|
||||
from src.interactive_gate import (
|
||||
mark_browser_activity,
|
||||
maybe_stop_background_tasks_for_heartbeat,
|
||||
)
|
||||
|
||||
await mark_browser_activity()
|
||||
|
||||
async def _stop_background():
|
||||
try:
|
||||
await maybe_stop_background_tasks_for_heartbeat(
|
||||
task_scheduler.stop_background_tasks_for_foreground
|
||||
)
|
||||
except Exception:
|
||||
logging.getLogger("app.foreground_gate").debug(
|
||||
"heartbeat task stop failed",
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
asyncio.create_task(_stop_background())
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
# Uploads
|
||||
from routes.upload_routes import setup_upload_routes
|
||||
upload_router, upload_cleanup_func = setup_upload_routes(upload_handler)
|
||||
@@ -597,14 +682,20 @@ app.include_router(setup_emoji_routes())
|
||||
# Sessions
|
||||
from routes.session_routes import setup_session_routes
|
||||
session_config = {"REQUEST_TIMEOUT": REQUEST_TIMEOUT, "OPENAI_API_KEY": OPENAI_API_KEY, "SESSIONS_FILE": SESSIONS_FILE}
|
||||
app.include_router(setup_session_routes(session_manager, session_config, webhook_manager=webhook_manager))
|
||||
app.include_router(setup_session_routes(
|
||||
session_manager,
|
||||
session_config,
|
||||
webhook_manager=webhook_manager,
|
||||
upload_handler=upload_handler,
|
||||
skills_manager=skills_manager,
|
||||
))
|
||||
|
||||
# Admin Danger Zone wipes (Settings → System → Danger Zone)
|
||||
from routes.admin_wipe_routes import setup_admin_wipe_routes
|
||||
from routes.admin_wipe.admin_wipe_routes import setup_admin_wipe_routes
|
||||
app.include_router(setup_admin_wipe_routes(session_manager))
|
||||
|
||||
# Memory
|
||||
from routes.memory_routes import setup_memory_routes
|
||||
from routes.memory.memory_routes import setup_memory_routes
|
||||
memory_router = setup_memory_routes(memory_manager, session_manager, memory_vector=memory_vector)
|
||||
app.include_router(memory_router)
|
||||
from routes.skills_routes import setup_skills_routes
|
||||
@@ -625,11 +716,11 @@ from routes.research.research_routes import setup_research_routes
|
||||
app.include_router(setup_research_routes(research_handler, session_manager=session_manager))
|
||||
|
||||
# History
|
||||
from routes.history_routes import setup_history_routes
|
||||
app.include_router(setup_history_routes(session_manager))
|
||||
from routes.history.history_routes import setup_history_routes
|
||||
app.include_router(setup_history_routes(session_manager, upload_handler=upload_handler))
|
||||
|
||||
# Search
|
||||
from routes.search_routes import setup_search_routes
|
||||
from routes.search.search_routes import setup_search_routes
|
||||
app.include_router(setup_search_routes(config))
|
||||
|
||||
# Presets
|
||||
@@ -641,7 +732,7 @@ from routes.diagnostics_routes import setup_diagnostics_routes
|
||||
app.include_router(setup_diagnostics_routes(rag_manager, rag_available, research_handler, memory_vector))
|
||||
|
||||
# Cleanup
|
||||
from routes.cleanup_routes import setup_cleanup_routes
|
||||
from routes.cleanup.cleanup_routes import setup_cleanup_routes
|
||||
app.include_router(setup_cleanup_routes(session_manager))
|
||||
|
||||
# Personal docs
|
||||
@@ -676,7 +767,7 @@ app.include_router(setup_stt_routes(stt_service))
|
||||
logger.info("STT service initialized (provider managed via settings)")
|
||||
|
||||
# Documents (artifacts/canvas)
|
||||
from routes.document_routes import setup_document_routes
|
||||
from routes.document.document_routes import setup_document_routes
|
||||
document_router = setup_document_routes(session_manager, upload_handler)
|
||||
app.include_router(document_router)
|
||||
|
||||
@@ -697,7 +788,7 @@ from src.task_scheduler import TaskScheduler
|
||||
task_scheduler = TaskScheduler(session_manager)
|
||||
from src.event_bus import set_task_scheduler
|
||||
set_task_scheduler(task_scheduler)
|
||||
from routes.task_routes import setup_task_routes
|
||||
from routes.task.task_routes import setup_task_routes
|
||||
app.include_router(setup_task_routes(task_scheduler))
|
||||
|
||||
from routes.assistant_routes import setup_assistant_routes
|
||||
@@ -705,7 +796,7 @@ app.include_router(setup_assistant_routes(task_scheduler))
|
||||
|
||||
# Calendar (CalDAV)
|
||||
from routes.calendar_routes import setup_calendar_routes
|
||||
calendar_router = setup_calendar_routes()
|
||||
calendar_router = setup_calendar_routes(upload_handler=upload_handler)
|
||||
app.include_router(calendar_router)
|
||||
|
||||
# Shell (user-facing command execution)
|
||||
@@ -724,7 +815,7 @@ from routes.hwfit_routes import setup_hwfit_routes
|
||||
app.include_router(setup_hwfit_routes())
|
||||
|
||||
# Model A/B Comparison
|
||||
from routes.compare_routes import setup_compare_routes
|
||||
from routes.compare.compare_routes import setup_compare_routes
|
||||
app.include_router(setup_compare_routes(session_manager))
|
||||
|
||||
# User Preferences
|
||||
@@ -742,7 +833,7 @@ app.include_router(setup_font_routes())
|
||||
# MCP (Model Context Protocol)
|
||||
from src.mcp_manager import McpManager
|
||||
from src.agent_tools import set_mcp_manager
|
||||
from routes.mcp_routes import setup_mcp_routes
|
||||
from routes.mcp.mcp_routes import setup_mcp_routes
|
||||
|
||||
mcp_manager = McpManager()
|
||||
set_mcp_manager(mcp_manager)
|
||||
@@ -757,7 +848,7 @@ set_ai_rag_manager(rag_manager, personal_docs_mgr)
|
||||
logger.info("AI interaction tools initialized (session, memory, RAG, UI control)")
|
||||
|
||||
# Webhooks
|
||||
from routes.webhook_routes import setup_webhook_routes
|
||||
from routes.webhook.webhook_routes import setup_webhook_routes
|
||||
app.include_router(setup_webhook_routes(webhook_manager, auth_manager, session_manager, api_key_manager))
|
||||
|
||||
# API Tokens
|
||||
@@ -767,8 +858,8 @@ app.include_router(setup_api_token_routes())
|
||||
logger.info("Webhook & API token routes initialized")
|
||||
|
||||
# Notes (Google Keep-style notes/todos)
|
||||
from routes.note_routes import setup_note_routes
|
||||
app.include_router(setup_note_routes(task_scheduler))
|
||||
from routes.note.note_routes import setup_note_routes
|
||||
app.include_router(setup_note_routes(task_scheduler, upload_handler=upload_handler))
|
||||
|
||||
# Email
|
||||
from routes.email_routes import setup_email_routes
|
||||
@@ -789,11 +880,11 @@ app.include_router(setup_codex_routes(
|
||||
))
|
||||
app.include_router(setup_claude_routes())
|
||||
|
||||
from routes.vault_routes import setup_vault_routes
|
||||
from routes.vault.vault_routes import setup_vault_routes
|
||||
app.include_router(setup_vault_routes())
|
||||
|
||||
# Contacts (CardDAV)
|
||||
from routes.contacts_routes import setup_contacts_routes
|
||||
from routes.contacts.contacts_routes import setup_contacts_routes
|
||||
app.include_router(setup_contacts_routes())
|
||||
|
||||
from companion import setup_companion_routes
|
||||
@@ -862,13 +953,45 @@ async def serve_login(request: Request):
|
||||
|
||||
@app.get("/api/version")
|
||||
async def get_version():
|
||||
from core.constants import APP_VERSION
|
||||
return {"version": APP_VERSION}
|
||||
from core.constants import APP_BUILD_VERSION, APP_SOURCE_COMMIT, APP_VERSION
|
||||
return {
|
||||
"version": APP_VERSION,
|
||||
"build": APP_BUILD_VERSION,
|
||||
"source_commit": APP_SOURCE_COMMIT,
|
||||
}
|
||||
|
||||
@app.get("/api/health")
|
||||
async def health_check() -> Dict[str, str]:
|
||||
return {"status": "healthy", "timestamp": datetime.now(timezone.utc).isoformat()}
|
||||
|
||||
@app.post("/api/client-perf")
|
||||
async def client_perf(request: Request):
|
||||
"""Low-volume frontend timing reports for stalls that happen before SSE logs."""
|
||||
try:
|
||||
data = await request.json()
|
||||
except Exception:
|
||||
data = {}
|
||||
try:
|
||||
kind = str(data.get("type") or "client").replace("\n", " ")[:80]
|
||||
total_ms = float(data.get("total_ms") or 0)
|
||||
stages = data.get("stages") if isinstance(data.get("stages"), list) else []
|
||||
stage_txt = " ".join(
|
||||
f"{str(s.get('name') or '')[:40]}={float(s.get('delta_ms') or 0):.0f}ms"
|
||||
for s in stages[:20]
|
||||
if isinstance(s, dict)
|
||||
)
|
||||
extra = str(data.get("extra") or "").replace("\n", " ")[:200]
|
||||
logging.getLogger("app.client_perf").warning(
|
||||
"client_perf type=%s total=%.0fms %s%s",
|
||||
kind,
|
||||
total_ms,
|
||||
stage_txt,
|
||||
f" extra={extra}" if extra else "",
|
||||
)
|
||||
except Exception:
|
||||
logging.getLogger("app.client_perf").debug("client_perf log failed", exc_info=True)
|
||||
return {"ok": True}
|
||||
|
||||
@app.get("/api/ready")
|
||||
async def readiness_check() -> JSONResponse:
|
||||
"""Readiness / integrity self-check — DB, data dir, local-first storage.
|
||||
@@ -895,11 +1018,76 @@ async def runtime_info() -> Dict[str, object]:
|
||||
or os.getenv("OLLAMA_URL")
|
||||
or ("http://host.docker.internal:11434/v1" if in_docker else "http://127.0.0.1:11434/v1")
|
||||
)
|
||||
network_mode = os.getenv("ODYSSEUS_CONTAINER_NETWORK_MODE", "").strip()
|
||||
host_gateway_reachable = False
|
||||
host_gateway_address = ""
|
||||
if in_docker and network_mode != "host":
|
||||
try:
|
||||
resolved = socket.getaddrinfo("host.docker.internal", None)
|
||||
for item in resolved:
|
||||
sockaddr = item[4] if len(item) >= 5 else ()
|
||||
candidate = sockaddr[0] if sockaddr else ""
|
||||
if candidate:
|
||||
host_gateway_address = str(candidate)
|
||||
break
|
||||
host_gateway_reachable = True
|
||||
except OSError:
|
||||
host_gateway_reachable = False
|
||||
if not host_gateway_address:
|
||||
host_gateway_address = _docker_default_gateway_ip()
|
||||
container: Dict[str, object] = {
|
||||
"engine": "docker" if in_docker else "",
|
||||
"networkMode": network_mode,
|
||||
"hostAccess": bool(in_docker and network_mode == "host"),
|
||||
"hostGatewayReachable": host_gateway_reachable,
|
||||
}
|
||||
if host_gateway_address:
|
||||
container["hostGatewayAddress"] = host_gateway_address
|
||||
command_names = (
|
||||
"ip",
|
||||
"ss",
|
||||
"arp",
|
||||
"nmap",
|
||||
"ping",
|
||||
"dig",
|
||||
"ssh",
|
||||
"git",
|
||||
"docker",
|
||||
)
|
||||
commands = {name: bool(shutil.which(name)) for name in command_names}
|
||||
capabilities = {
|
||||
"networkInspection": bool(commands["ip"] and (commands["ss"] or commands["arp"])),
|
||||
"lanScan": bool(commands["nmap"]),
|
||||
"dnsLookup": bool(commands["dig"]),
|
||||
"sshClient": bool(commands["ssh"]),
|
||||
"git": bool(commands["git"]),
|
||||
"dockerClient": bool(commands["docker"]),
|
||||
}
|
||||
return {
|
||||
"in_docker": in_docker,
|
||||
"ollama_base_url": ollama_url,
|
||||
"container": container,
|
||||
"commands": commands,
|
||||
"capabilities": capabilities,
|
||||
}
|
||||
|
||||
|
||||
def _docker_default_gateway_ip() -> str:
|
||||
try:
|
||||
with open("/proc/net/route", "r", encoding="utf-8", errors="ignore") as fh:
|
||||
for line in fh.readlines()[1:]:
|
||||
parts = line.split()
|
||||
if len(parts) < 3 or parts[1] != "00000000":
|
||||
continue
|
||||
raw = parts[2]
|
||||
if len(raw) != 8:
|
||||
continue
|
||||
octets = [str(int(raw[i:i + 2], 16)) for i in range(6, -1, -2)]
|
||||
return ".".join(octets)
|
||||
except Exception:
|
||||
return ""
|
||||
return ""
|
||||
|
||||
# ========= LIFECYCLE =========
|
||||
|
||||
@asynccontextmanager
|
||||
@@ -939,6 +1127,15 @@ async def _startup_event():
|
||||
# GC tasks created with `asyncio.create_task(...)` before they finish.
|
||||
_startup_tasks: list[asyncio.Task] = getattr(app.state, "_startup_tasks", [])
|
||||
app.state._startup_tasks = _startup_tasks
|
||||
from src.background_tool_jobs import BackgroundToolJobs
|
||||
from routes.chat_routes import _active_streams
|
||||
from src import agent_runs
|
||||
app.state.background_tool_jobs = BackgroundToolJobs(
|
||||
is_busy=lambda sid: sid in _active_streams or agent_runs.is_active(sid),
|
||||
session_manager=session_manager, research_handler=research_handler,
|
||||
)
|
||||
app.state.background_tool_delivery_task = asyncio.create_task(app.state.background_tool_jobs.run())
|
||||
_startup_tasks.append(app.state.background_tool_delivery_task)
|
||||
if upload_cleanup_func:
|
||||
upload_cleanup_task = asyncio.create_task(upload_cleanup_func())
|
||||
# Always-on monitor that auto-continues the agent when a background bash
|
||||
@@ -957,7 +1154,7 @@ async def _startup_event():
|
||||
except BaseException as e:
|
||||
logger.warning(f"Built-in MCP registration failed (non-critical): {type(e).__name__}: {e}")
|
||||
try:
|
||||
await asyncio.wait_for(mcp_manager.connect_all_enabled(), timeout=20)
|
||||
await mcp_manager.connect_all_enabled()
|
||||
except asyncio.TimeoutError:
|
||||
logger.warning("User MCP startup timed out (non-critical)")
|
||||
except BaseException as e:
|
||||
@@ -965,57 +1162,70 @@ async def _startup_event():
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_startup_mcp_connections()))
|
||||
|
||||
# Pre-warm the RAG tool index off the request path. Loading the local
|
||||
# embedding model + opening ChromaDB + indexing the built-in tools is a
|
||||
# one-time ~1-3s cost that otherwise lands on the user's FIRST message
|
||||
# (showing up as a big `tool_selection` time). Doing it here makes the
|
||||
# first turn as fast as subsequent ones (warm embed ≈ a few ms).
|
||||
async def _warmup_tool_index():
|
||||
try:
|
||||
from src.tool_index import get_tool_index
|
||||
idx = await asyncio.to_thread(get_tool_index)
|
||||
if idx:
|
||||
await asyncio.to_thread(idx.get_tools_for_query, "warmup", 8)
|
||||
logger.info("[startup] Tool index pre-warmed")
|
||||
except Exception as e:
|
||||
logger.warning(f"Tool index warmup failed (non-critical): {type(e).__name__}: {e}")
|
||||
# Semantic tool selection is part of the agent serving contract. Initialize
|
||||
# it in a background thread by default so startup remains nonblocking while
|
||||
# harness deployments can wait for the explicit readiness state.
|
||||
from src.tool_index import prewarm_tool_index, tool_index_prewarm_enabled
|
||||
if tool_index_prewarm_enabled():
|
||||
async def _warmup_tool_index():
|
||||
status = await asyncio.to_thread(prewarm_tool_index)
|
||||
if status.get("ready"):
|
||||
logger.info(
|
||||
"[startup] Tool index pre-warmed lanes=%s tools=%s duration_ms=%s",
|
||||
[lane.get("name") for lane in status.get("lanes", [])],
|
||||
status.get("builtin_tools"),
|
||||
status.get("duration_ms"),
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
"Tool index warmup degraded (non-critical): %s",
|
||||
status.get("error_type") or status.get("state"),
|
||||
)
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_warmup_tool_index()))
|
||||
# Warmup: ping all known LLM endpoints to prime connections
|
||||
async def _warmup_endpoints():
|
||||
try:
|
||||
import httpx
|
||||
# model_discovery has no get_endpoints(); that call raised
|
||||
# AttributeError every run and silently disabled warmup/keepalive.
|
||||
# Resolve the /models probe URLs via the real discovery API, off the
|
||||
# event loop since discovery does a blocking port scan.
|
||||
urls = (
|
||||
await asyncio.to_thread(model_discovery.warmup_ping_urls)
|
||||
if model_discovery else []
|
||||
)
|
||||
for url in urls:
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||
await client.get(url)
|
||||
logger.info(f"Warmup ping OK: {url}")
|
||||
except Exception as e:
|
||||
logger.debug(f"Warmup ping failed for endpoint: {e}")
|
||||
except Exception as e:
|
||||
logger.debug(f"Warmup ping skipped: {e}")
|
||||
_startup_tasks.append(asyncio.create_task(_warmup_tool_index()))
|
||||
else:
|
||||
logger.info("Tool index prewarm disabled (ODYSSEUS_TOOL_INDEX_PREWARM=0)")
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_warmup_endpoints()))
|
||||
|
||||
# Keep-alive: ping endpoints every 60 seconds to prevent cold starts
|
||||
async def _keepalive_loop():
|
||||
while True:
|
||||
# Model endpoint pings remain opt-in. They can compete with the first seconds
|
||||
# of UI use on slow or busy machines and are not required for local startup.
|
||||
_startup_warmups_enabled = str(os.getenv("ODYSSEUS_STARTUP_WARMUPS", "")).lower() in {"1", "true", "yes", "on"}
|
||||
if _startup_warmups_enabled:
|
||||
async def _warmup_endpoints():
|
||||
try:
|
||||
await asyncio.sleep(60)
|
||||
await _warmup_endpoints()
|
||||
import httpx
|
||||
urls = (
|
||||
await asyncio.to_thread(model_discovery.warmup_ping_urls)
|
||||
if model_discovery else []
|
||||
)
|
||||
for url in urls:
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||
await client.get(url)
|
||||
logger.info(f"Warmup ping OK: {url}")
|
||||
except Exception as e:
|
||||
logger.debug(f"Warmup ping failed for endpoint: {e}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Keepalive loop error: {e}")
|
||||
await asyncio.sleep(300) # Back off on error
|
||||
logger.debug(f"Warmup ping skipped: {e}")
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_keepalive_loop()))
|
||||
_startup_tasks.append(asyncio.create_task(_warmup_endpoints()))
|
||||
else:
|
||||
logger.info("Model endpoint warmups disabled (set ODYSSEUS_STARTUP_WARMUPS=1 to enable)")
|
||||
|
||||
# Keep-alive is opt-in. The ping path performs model discovery, and when
|
||||
# stale LAN endpoints are configured it can add periodic backend pressure
|
||||
# that delays unrelated UI requests such as Notes/Documents.
|
||||
_keepalive_enabled = str(os.getenv("ODYSSEUS_MODEL_KEEPALIVE", "")).lower() in {"1", "true", "yes", "on"}
|
||||
if _keepalive_enabled:
|
||||
async def _keepalive_loop():
|
||||
while True:
|
||||
try:
|
||||
await asyncio.sleep(60)
|
||||
await _warmup_endpoints()
|
||||
except Exception as e:
|
||||
logger.warning(f"Keepalive loop error: {e}")
|
||||
await asyncio.sleep(300) # Back off on error
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_keepalive_loop()))
|
||||
|
||||
async def _ensure_default_tasks():
|
||||
# Create/reconcile default automation tasks + personal assistant for every user.
|
||||
@@ -1067,6 +1277,14 @@ async def _startup_event():
|
||||
# Disk-backed skills are not covered by the DB legacy-owner sweep. Repair
|
||||
# ownerless or deleted/test-owner SKILL.md files so strict owner filtering
|
||||
# does not make an existing library look empty after auth/account changes.
|
||||
try:
|
||||
from services.memory.builtin_skills import install_builtin_skills
|
||||
installed = install_builtin_skills(skills_manager, ())
|
||||
if installed:
|
||||
logger.info("Installed %s built-in skill file(s)", installed)
|
||||
except Exception as e:
|
||||
logger.debug(f"Built-in skill installation skipped: {e}")
|
||||
|
||||
try:
|
||||
import json as _json
|
||||
auth_path = AUTH_FILE
|
||||
@@ -1112,35 +1330,10 @@ async def _startup_event():
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_null_owner_sweep_loop()))
|
||||
|
||||
# Nightly skill audit — at ~02:00 local, test + judge a batch of the
|
||||
# least-recently-checked skills, auto-fixing/escalating weak ones (never
|
||||
# deletes). Rotates through the library so each night covers different
|
||||
# skills. Gated by the `skill_audit_nightly` setting (default on); hour via
|
||||
# `skill_audit_hour` (default 2), batch size via `skill_audit_batch` (8).
|
||||
async def _skill_audit_nightly_loop():
|
||||
from datetime import timedelta
|
||||
while True:
|
||||
try:
|
||||
from src.settings import get_setting
|
||||
hour = int(get_setting("skill_audit_hour", 2) or 2)
|
||||
except Exception:
|
||||
hour = 2
|
||||
now = datetime.now()
|
||||
nxt = now.replace(hour=hour % 24, minute=0, second=0, microsecond=0)
|
||||
if nxt <= now:
|
||||
nxt += timedelta(days=1)
|
||||
await asyncio.sleep(max(60, (nxt - now).total_seconds()))
|
||||
try:
|
||||
from src.settings import get_setting
|
||||
if not get_setting("skill_audit_nightly", True):
|
||||
continue
|
||||
batch = int(get_setting("skill_audit_batch", 8) or 8)
|
||||
from routes.skills_routes import run_scheduled_skill_audit
|
||||
await run_scheduled_skill_audit(skills_manager, owner=None, max_skills=batch)
|
||||
except Exception as e:
|
||||
logger.warning(f"Nightly skill audit failed: {e}")
|
||||
|
||||
_startup_tasks.append(asyncio.create_task(_skill_audit_nightly_loop()))
|
||||
# Skills Audit is scheduled per owner by TaskScheduler. Do not also start
|
||||
# an ownerless audit here: its sidecar results cannot be read back through
|
||||
# an authenticated owner's skill namespace, and its model activity can
|
||||
# defer the real per-owner task at the same time of night.
|
||||
|
||||
# Cookbook serve lifecycle — kills scheduler-launched serves whose
|
||||
# window-end has passed. Paired with the cookbook_serve builtin
|
||||
@@ -1155,6 +1348,18 @@ async def _startup_event():
|
||||
|
||||
async def _shutdown_event():
|
||||
logger.info("Application shutting down...")
|
||||
background_delivery = getattr(app.state, 'background_tool_delivery_task', None)
|
||||
if background_delivery:
|
||||
background_delivery.cancel()
|
||||
try:
|
||||
await background_delivery
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
try:
|
||||
from src.agent_tools.web_tools import shutdown_private_browser_sessions
|
||||
await shutdown_private_browser_sessions()
|
||||
except Exception as e:
|
||||
logger.warning(f"Private browser shutdown error: {e}")
|
||||
if upload_cleanup_task:
|
||||
upload_cleanup_task.cancel()
|
||||
try:
|
||||
@@ -1183,6 +1388,6 @@ if __name__ == "__main__":
|
||||
import uvicorn
|
||||
|
||||
bind_host = os.getenv("APP_BIND", "127.0.0.1")
|
||||
bind_port = int(os.getenv("APP_PORT", "7000"))
|
||||
bind_port = int(os.getenv("APP_PORT", "7011"))
|
||||
|
||||
uvicorn.run(app, host=bind_host, port=bind_port, log_level="info")
|
||||
|
||||
|
Before Width: | Height: | Size: 185 KiB After Width: | Height: | Size: 185 KiB |
|
Before Width: | Height: | Size: 16 KiB After Width: | Height: | Size: 16 KiB |
|
Before Width: | Height: | Size: 79 KiB After Width: | Height: | Size: 79 KiB |
+8
-4
@@ -27,13 +27,13 @@ echo " port: $PORT"
|
||||
rm -rf "$APP"
|
||||
mkdir -p "$APP/Contents/MacOS" "$APP/Contents/Resources"
|
||||
|
||||
# ── Icon (best effort) — center-crop docs/odysseus.jpg to a square .icns ──
|
||||
if [ -f "$REPO_DIR/docs/odysseus.jpg" ] && command -v sips >/dev/null 2>&1; then
|
||||
# ── Icon (best effort) — center-crop the branding image to a square .icns ──
|
||||
if [ -f "$REPO_DIR/assets/branding/odysseus.jpg" ] && command -v sips >/dev/null 2>&1; then
|
||||
TMPIMG="$(mktemp -d)"
|
||||
# Center-crop to a square, scale to 512 (sips' icns encoder caps at 512), and
|
||||
# let sips emit the .icns directly — more robust across macOS versions than
|
||||
# building an .iconset by hand.
|
||||
sips -c 720 720 "$REPO_DIR/docs/odysseus.jpg" --out "$TMPIMG/sq.png" >/dev/null 2>&1 || cp "$REPO_DIR/docs/odysseus.jpg" "$TMPIMG/sq.png"
|
||||
sips -c 720 720 "$REPO_DIR/assets/branding/odysseus.jpg" --out "$TMPIMG/sq.png" >/dev/null 2>&1 || cp "$REPO_DIR/assets/branding/odysseus.jpg" "$TMPIMG/sq.png"
|
||||
sips -z 512 512 "$TMPIMG/sq.png" --out "$TMPIMG/icon.png" >/dev/null 2>&1
|
||||
if sips -s format icns "$TMPIMG/icon.png" --out "$APP/Contents/Resources/odysseus.icns" >/dev/null 2>&1; then
|
||||
echo " icon: odysseus.icns"
|
||||
@@ -42,7 +42,7 @@ if [ -f "$REPO_DIR/docs/odysseus.jpg" ] && command -v sips >/dev/null 2>&1; then
|
||||
fi
|
||||
rm -rf "$TMPIMG"
|
||||
else
|
||||
echo " icon: (skipped — no docs/odysseus.jpg)"
|
||||
echo " icon: (skipped — no assets/branding/odysseus.jpg)"
|
||||
fi
|
||||
|
||||
# ── Info.plist ──
|
||||
@@ -73,6 +73,10 @@ cat > "$APP/Contents/MacOS/$APP_NAME.tmpl" <<'LAUNCHER'
|
||||
INSTALL_DIR="__INSTALL_DIR__"
|
||||
PORT="__PORT__"
|
||||
URL="http://127.0.0.1:${PORT}"
|
||||
# uvicorn is started with --port below, but APP_PORT is what the app itself
|
||||
# reads when it needs to build a URL for this instance (internal_api_base(),
|
||||
# companion pairing, the MCP OAuth callback), so export it as well.
|
||||
export APP_PORT="$PORT"
|
||||
export PATH="/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:$PATH"
|
||||
|
||||
UVICORN="$INSTALL_DIR/venv/bin/uvicorn"
|
||||
|
||||
@@ -6,11 +6,14 @@ units so the route layer stays thin and the logic is directly testable.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import secrets
|
||||
import socket
|
||||
import uuid
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import bcrypt
|
||||
|
||||
@@ -20,6 +23,102 @@ PAIRING_VERSION = 1
|
||||
COMPANION_SCOPE = "chat"
|
||||
|
||||
|
||||
_COMPANION_IPV4_NETWORKS = tuple(
|
||||
ipaddress.ip_network(cidr)
|
||||
for cidr in (
|
||||
"10.0.0.0/8",
|
||||
"100.64.0.0/10",
|
||||
"127.0.0.0/8",
|
||||
"169.254.0.0/16",
|
||||
"172.16.0.0/12",
|
||||
"192.168.0.0/16",
|
||||
)
|
||||
)
|
||||
_DNS_LABEL_RE = re.compile(r"[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\Z")
|
||||
|
||||
|
||||
def _valid_companion_client_host(host: str) -> bool:
|
||||
"""Match the host forms supported by the current v1 Expo client."""
|
||||
if not host or len(host) > 253 or not host.isascii() or "%" in host:
|
||||
return False
|
||||
|
||||
try:
|
||||
address = ipaddress.ip_address(host)
|
||||
except ValueError:
|
||||
labels = host.split(".")
|
||||
if any(not _DNS_LABEL_RE.fullmatch(label) for label in labels):
|
||||
return False
|
||||
if any(label.startswith("xn--") for label in labels):
|
||||
return False
|
||||
# WHATWG URL parsers treat a decimal or ``0x`` single-label hostname
|
||||
# as an IPv4 number even though Python's strict ``ipaddress`` parser
|
||||
# rejects that spelling. The v1 client interpolates this host back
|
||||
# into a URL, so accepting e.g. ``134744072`` would make the phone send
|
||||
# its bearer token to public 8.8.8.8. Keep DNS labels unambiguous.
|
||||
if len(labels) == 1 and (
|
||||
labels[0].isdigit()
|
||||
or re.fullmatch(r"0x[0-9a-f]*", labels[0]) is not None
|
||||
):
|
||||
return False
|
||||
return len(labels) == 1 or (len(labels) >= 2 and labels[-1] == "local")
|
||||
|
||||
return isinstance(address, ipaddress.IPv4Address) and any(
|
||||
address in network for network in _COMPANION_IPV4_NETWORKS
|
||||
)
|
||||
|
||||
|
||||
def parse_companion_base_url(value: str) -> tuple[str, int]:
|
||||
"""Validate a v1 companion address and return its legacy (host, port).
|
||||
|
||||
The deployed client understands only HTTP plus a LAN-style host and port.
|
||||
Reject anything outside that exact contract instead of advertising a URL
|
||||
the client would reject, downgrade, or interpret differently.
|
||||
"""
|
||||
if not isinstance(value, str) or not value:
|
||||
raise ValueError("COMPANION_BASE_URL must be a canonical HTTP LAN origin")
|
||||
if not value.isascii():
|
||||
raise ValueError("COMPANION_BASE_URL must contain only ASCII characters")
|
||||
if any(
|
||||
ord(char) <= 32 or ord(char) == 127 or char in {"\\", "%"}
|
||||
for char in value
|
||||
):
|
||||
raise ValueError(
|
||||
"COMPANION_BASE_URL contains a forbidden character"
|
||||
)
|
||||
|
||||
try:
|
||||
parsed = urlsplit(value)
|
||||
port = parsed.port
|
||||
except ValueError as exc:
|
||||
raise ValueError("COMPANION_BASE_URL must be a valid HTTP LAN origin") from exc
|
||||
|
||||
host = parsed.hostname
|
||||
if parsed.scheme.lower() != "http" or not parsed.netloc or not host:
|
||||
raise ValueError("COMPANION_BASE_URL must be a canonical HTTP LAN origin")
|
||||
if parsed.username is not None or parsed.password is not None:
|
||||
raise ValueError("COMPANION_BASE_URL must not contain credentials")
|
||||
if parsed.path or parsed.query or parsed.fragment:
|
||||
raise ValueError("COMPANION_BASE_URL must not contain a path, query, or fragment")
|
||||
if port is not None and not 1 <= port <= 65535:
|
||||
raise ValueError("COMPANION_BASE_URL port must be between 1 and 65535")
|
||||
if not _valid_companion_client_host(host):
|
||||
raise ValueError("COMPANION_BASE_URL host is not supported by companion v1")
|
||||
|
||||
netloc = f"{host}:{port}" if port is not None else host
|
||||
origin = f"http://{netloc}"
|
||||
if value != origin:
|
||||
raise ValueError("COMPANION_BASE_URL must be a canonical HTTP LAN origin")
|
||||
return host, port or 80
|
||||
|
||||
|
||||
def configured_companion_origin() -> tuple[str, int] | None:
|
||||
"""Return the validated operator-configured v1 address, if any."""
|
||||
value = os.environ.get("COMPANION_BASE_URL")
|
||||
if value is None or value == "":
|
||||
return None
|
||||
return parse_companion_base_url(value)
|
||||
|
||||
|
||||
def default_port() -> int:
|
||||
"""Best guess at the port the server is reachable on. Callers that know the
|
||||
real request port should pass it explicitly."""
|
||||
|
||||
+23
-8
@@ -23,7 +23,7 @@ from fastapi import APIRouter, HTTPException, Request
|
||||
from fastapi.responses import HTMLResponse
|
||||
|
||||
from core.middleware import require_admin
|
||||
from src.auth_helpers import get_current_user
|
||||
from src.auth_helpers import _auth_disabled, get_current_user
|
||||
|
||||
from companion import pairing as _pairing
|
||||
|
||||
@@ -113,8 +113,9 @@ def setup_companion_routes() -> APIRouter:
|
||||
The stock /api/models route scopes to get_current_user, which for a
|
||||
bearer token is the sandboxed pseudo-user "api" (owns nothing). Here we
|
||||
scope to the token's real owner instead, plus legacy null-owner shared
|
||||
rows -- the same rule as owner_filter. Read-only; never returns api_key
|
||||
material.
|
||||
rows -- the same rule as owner_filter. Explicit auth-disabled mode keeps
|
||||
the stock route's single-user all-endpoints view. Read-only; never
|
||||
returns api_key material.
|
||||
"""
|
||||
require_models_scope(request)
|
||||
import json as _json
|
||||
@@ -123,6 +124,11 @@ def setup_companion_routes() -> APIRouter:
|
||||
from src.endpoint_resolver import build_chat_url
|
||||
|
||||
owner = token_owner(request)
|
||||
single_user_mode = (
|
||||
owner is None
|
||||
and not getattr(request.state, "api_token", False)
|
||||
and _auth_disabled()
|
||||
)
|
||||
out = []
|
||||
db = SessionLocal()
|
||||
try:
|
||||
@@ -133,7 +139,7 @@ def setup_companion_routes() -> APIRouter:
|
||||
if owner:
|
||||
q = q.filter((ModelEndpoint.owner == owner) | (ModelEndpoint.owner == None)) # noqa: E711
|
||||
for ep in q.all():
|
||||
if not owner_can_see(ep.owner, owner):
|
||||
if not single_user_mode and not owner_can_see(ep.owner, owner):
|
||||
continue
|
||||
try:
|
||||
model_ids = _json.loads(ep.cached_models) if ep.cached_models else []
|
||||
@@ -194,19 +200,27 @@ def setup_companion_routes() -> APIRouter:
|
||||
the code works immediately, no restart. `?format=json` returns the
|
||||
payload for an in-app pairing screen."""
|
||||
require_admin(request)
|
||||
try:
|
||||
configured_origin = _pairing.configured_companion_origin()
|
||||
except ValueError as exc:
|
||||
raise HTTPException(500, str(exc)) from None
|
||||
owner = get_current_user(request)
|
||||
invalidate = getattr(request.app.state, "invalidate_token_cache", None)
|
||||
token_id, raw_token = mint_pairing_token(owner, invalidate)
|
||||
|
||||
hosts = _pairing.lan_ip_candidates()
|
||||
host = hosts[0] if hosts else "127.0.0.1"
|
||||
port = request.url.port or _pairing.default_port()
|
||||
if configured_origin:
|
||||
host, port = configured_origin
|
||||
hosts = [host]
|
||||
else:
|
||||
hosts = _pairing.lan_ip_candidates()
|
||||
host = hosts[0] if hosts else "127.0.0.1"
|
||||
port = request.url.port or _pairing.default_port()
|
||||
payload = _pairing.pairing_payload(host, port, raw_token)
|
||||
qr = _pairing.pairing_qr_png_data_uri(payload)
|
||||
qr_ok = bool(qr and qr.startswith("data:image/png;base64,"))
|
||||
|
||||
if (request.query_params.get("format") or "").lower() == "json":
|
||||
return {
|
||||
response = {
|
||||
"host": host,
|
||||
"port": port,
|
||||
"token": raw_token,
|
||||
@@ -215,6 +229,7 @@ def setup_companion_routes() -> APIRouter:
|
||||
"payload": payload,
|
||||
"qr": qr if qr_ok else None,
|
||||
}
|
||||
return response
|
||||
|
||||
import json as _json
|
||||
payload_json = _json.dumps(payload, separators=(",", ":"))
|
||||
|
||||
+38
-14
@@ -15,29 +15,53 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import uuid
|
||||
from typing import Any, Optional
|
||||
|
||||
|
||||
def atomic_write_json(path: str, data: Any, *, indent: Optional[int] = None) -> None:
|
||||
"""Atomically persist `data` as JSON at `path`.
|
||||
|
||||
The temp file uses the live PID as a suffix so two processes saving the
|
||||
same file (e.g. unit tests) don't collide on the rename target.
|
||||
The temp file uses a random suffix so two concurrent writers saving the
|
||||
same file don't collide on the rename target. A PID suffix does not do
|
||||
this: the PID is constant for the life of a process, so two writers on
|
||||
the same path within one process (or one single-process container, where
|
||||
the PID never changes at all) still race for the same temp file.
|
||||
"""
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
tmp = f"{path}.tmp.{os.getpid()}"
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=indent)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.replace(tmp, path)
|
||||
tmp = f"{path}.tmp.{uuid.uuid4().hex}"
|
||||
|
||||
try:
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=indent)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.replace(tmp, path)
|
||||
finally:
|
||||
# Directly unlink to avoid a check-then-act race condition.
|
||||
# Swallows FileNotFoundError (on success path) and other cleanup OSErrors.
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def atomic_write_text(path: str, text: str) -> None:
|
||||
if not isinstance(text, str):
|
||||
raise TypeError("atomic_write_text expects a string")
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
tmp = f"{path}.tmp.{os.getpid()}"
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.replace(tmp, path)
|
||||
tmp = f"{path}.tmp.{uuid.uuid4().hex}"
|
||||
|
||||
try:
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.replace(tmp, path)
|
||||
finally:
|
||||
# Directly unlink to avoid a check-then-act race condition.
|
||||
# Swallows FileNotFoundError (on success path) and other cleanup OSErrors.
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
+9
-16
@@ -20,7 +20,6 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
from core.atomic_io import atomic_write_json as _atomic_write_json # noqa: E402
|
||||
from core.middleware import INTERNAL_TOOL_USER # noqa: E402
|
||||
|
||||
DEFAULT_PRIVILEGES = {
|
||||
"can_use_agent": True,
|
||||
@@ -49,24 +48,18 @@ ADMIN_PRIVILEGES["allowed_models_restricted"] = False
|
||||
ADMIN_PRIVILEGES["block_all_models"] = False
|
||||
|
||||
from src.constants import AUTH_FILE, PASSWORD_MIN_LENGTH
|
||||
from src.owner_identity import RESERVED_AUTH_USERNAMES
|
||||
DEFAULT_AUTH_PATH = AUTH_FILE
|
||||
TOKEN_TTL = 60 * 60 * 24 * 7 # 7 days
|
||||
|
||||
# Usernames the auth + middleware layer reserve as internal "synthetic owner"
|
||||
# sentinels; they must never belong to a real account. The most dangerous is
|
||||
# "internal-tool": `core.middleware.require_admin` treats any request whose
|
||||
# `current_user == "internal-tool"` as the in-process tool loopback and grants
|
||||
# admin, and because the cookie auth path sets `current_user` to the raw
|
||||
# username, an account literally named "internal-tool" would be silently
|
||||
# treated as an admin by every `require_admin`-gated route. "api" collides with
|
||||
# the bearer-token owner-attribution sentinel. "demo"/"system" round out the
|
||||
# synthetic-owner set the rest of the codebase already special-cases (see
|
||||
# `_SYNTHETIC_OWNERS` in routes/assistant_routes.py and the matching guards in
|
||||
# src/task_scheduler.py / routes/research_routes.py) — a real account with one
|
||||
# of those names would be denied an assistant and inconsistently owner-scoped.
|
||||
# Refuse to create or rename into any of them so the sentinels can't be
|
||||
# impersonated. (Keep this in sync with that synthetic-owner set.)
|
||||
RESERVED_USERNAMES = frozenset({INTERNAL_TOOL_USER, "api", "demo", "system"})
|
||||
# Usernames the auth + middleware layer reserves for request sentinels and
|
||||
# internal storage owners; they must never belong to a real login account.
|
||||
# "internal-tool" is the most dangerous because `core.middleware.require_admin`
|
||||
# treats it as the in-process tool loopback. "api" collides with bearer-token
|
||||
# attribution. "demo"/"system" are synthetic owners already special-cased by
|
||||
# scheduler/assistant/research paths. The Default/Local owner is a storage
|
||||
# bucket for explicit auth-disabled no-login mode, not a login username.
|
||||
RESERVED_USERNAMES = frozenset(RESERVED_AUTH_USERNAMES)
|
||||
|
||||
|
||||
def normalize_known_username(users: Dict[str, Any], username: str | None) -> Optional[str]:
|
||||
|
||||
+696
-81
File diff suppressed because it is too large
Load Diff
+30
-4
@@ -3,10 +3,14 @@
|
||||
|
||||
import os
|
||||
import secrets
|
||||
from collections.abc import Mapping
|
||||
|
||||
from fastapi import HTTPException, Request
|
||||
from starlette.middleware.base import BaseHTTPMiddleware
|
||||
from starlette.responses import Response
|
||||
from starlette.routing import get_route_path
|
||||
|
||||
from src.owner_identity import INTERNAL_TOOL_USER, auth_disabled
|
||||
|
||||
|
||||
# Per-process token that lets the in-app tool layer hit admin-gated
|
||||
@@ -15,8 +19,30 @@ from starlette.responses import Response
|
||||
# same value from this module. Never persisted or exposed externally.
|
||||
INTERNAL_TOOL_TOKEN = os.environ.get("ODYSSEUS_INTERNAL_TOKEN") or secrets.token_hex(32)
|
||||
INTERNAL_TOOL_HEADER = "X-Odysseus-Internal-Token"
|
||||
# Pseudo-username on in-process tool-loopback requests; require_admin trusts it and it is reserved.
|
||||
INTERNAL_TOOL_USER = "internal-tool"
|
||||
|
||||
|
||||
def get_application_route_path(scope: Mapping[str, object]) -> str:
|
||||
"""Return the application-relative path used by Starlette routing.
|
||||
|
||||
Uvicorn prefixes ``scope["path"]`` with a configured ASGI ``root_path``;
|
||||
Starlette removes that prefix before matching routes. Middleware policy
|
||||
must use the same path form or a deployment prefix can change which policy
|
||||
applies to an otherwise unchanged application route.
|
||||
"""
|
||||
return get_route_path(scope)
|
||||
|
||||
|
||||
def with_asgi_root_path(scope: Mapping[str, object], path: str) -> str:
|
||||
"""Prefix an application path for a client-facing redirect target."""
|
||||
root_path = scope.get("root_path", "")
|
||||
if not isinstance(root_path, str) or not root_path:
|
||||
return path
|
||||
return f"{root_path.rstrip('/')}{path}"
|
||||
|
||||
|
||||
def path_is_route_or_child(path: str, prefix: str) -> bool:
|
||||
"""Return whether ``path`` is exactly ``prefix`` or below that route."""
|
||||
return path == prefix or path.startswith(prefix + "/")
|
||||
|
||||
|
||||
def is_cors_preflight(method: str, headers) -> bool:
|
||||
@@ -47,7 +73,7 @@ def require_admin(request: Request):
|
||||
pass
|
||||
|
||||
auth_mgr = getattr(request.app.state, "auth_manager", None)
|
||||
if os.getenv("AUTH_ENABLED", "true").lower() == "false":
|
||||
if auth_disabled():
|
||||
return
|
||||
if not auth_mgr or not auth_mgr.is_configured:
|
||||
raise HTTPException(403, "Admin only")
|
||||
@@ -117,7 +143,7 @@ class SecurityHeadersMiddleware(BaseHTTPMiddleware):
|
||||
f"script-src 'self' 'nonce-{nonce}' https://cdn.jsdelivr.net; "
|
||||
"style-src 'self' 'unsafe-inline' https://cdn.jsdelivr.net; "
|
||||
"font-src 'self' https://cdn.jsdelivr.net; "
|
||||
"img-src 'self' data: blob:; "
|
||||
"img-src 'self' data: blob: https:; "
|
||||
"media-src 'self' blob:; "
|
||||
"connect-src 'self'; "
|
||||
"frame-src 'self'; "
|
||||
|
||||
+75
-1
@@ -8,6 +8,11 @@ These are simple datacontainers. All persistence is handled by SessionManager.
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Any, Optional, TYPE_CHECKING
|
||||
|
||||
from src.tool_approval_scopes import (
|
||||
CHAT_SESSION_APPROVAL_CONTEXT_MARKER,
|
||||
CHAT_SESSION_APPROVAL_DECISION,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from .session_manager import SessionManager
|
||||
|
||||
@@ -31,6 +36,35 @@ set_session_manager = set_session_manager_instance
|
||||
get_session_manager = get_session_manager_instance
|
||||
|
||||
|
||||
def _history_grants_chat_session_approval(
|
||||
history: List["ChatMessage"],
|
||||
session_id: str,
|
||||
) -> bool:
|
||||
"""Return whether this exact chat has a resolved session-scope grant."""
|
||||
|
||||
expected_session = str(session_id or "")
|
||||
if not expected_session:
|
||||
return False
|
||||
for message in reversed(history or []):
|
||||
metadata = getattr(message, "metadata", None)
|
||||
if not isinstance(metadata, dict):
|
||||
continue
|
||||
tool_events = metadata.get("tool_events")
|
||||
if not isinstance(tool_events, list):
|
||||
continue
|
||||
for event in reversed(tool_events):
|
||||
ask_user = event.get("ask_user") if isinstance(event, dict) else None
|
||||
if not isinstance(ask_user, dict):
|
||||
continue
|
||||
if (
|
||||
ask_user.get("kind") == "tool_approval"
|
||||
and ask_user.get("resolved") == CHAT_SESSION_APPROVAL_DECISION
|
||||
and str(ask_user.get("session_id") or "") == expected_session
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@dataclass
|
||||
class ChatMessage:
|
||||
"""A single chat message."""
|
||||
@@ -74,6 +108,12 @@ class Session:
|
||||
owner: Optional[str] = None
|
||||
is_important: bool = False
|
||||
message_count: int = 0
|
||||
memory_extraction_enabled: bool = True
|
||||
skill_injection_enabled: bool = True
|
||||
thinking_mode: str = "off"
|
||||
temperature_override: Optional[float] = None
|
||||
max_tokens_override: Optional[int] = None
|
||||
cwd: Optional[str] = None
|
||||
|
||||
def __post_init__(self):
|
||||
if self.headers is None:
|
||||
@@ -116,11 +156,45 @@ class Session:
|
||||
the model. Display/history-load paths use the raw ``history`` and are
|
||||
unaffected.
|
||||
"""
|
||||
return [
|
||||
messages = [
|
||||
msg.to_dict()
|
||||
for msg in self.history
|
||||
if (msg.metadata or {}).get("source") != "slash"
|
||||
]
|
||||
from src.background_tool_jobs import background_result_context
|
||||
messages = [part for message in messages for part in (
|
||||
*background_result_context(message.get('metadata')), message,
|
||||
)]
|
||||
# Resume an interrupted thinking-only response from its actual model
|
||||
# reasoning channel. Restrict this to the latest assistant message so
|
||||
# old traces do not accumulate in context or cause reasoning loops.
|
||||
for index in range(len(messages) - 1, -1, -1):
|
||||
message = messages[index]
|
||||
if message.get("role") != "assistant":
|
||||
continue
|
||||
metadata = message.get("metadata") or {}
|
||||
thinking = str(metadata.get("thinking") or "").strip()
|
||||
if metadata.get("stopped") and thinking:
|
||||
resumed = dict(message)
|
||||
resumed["reasoning_content"] = thinking
|
||||
messages[index] = resumed
|
||||
break
|
||||
if not _history_grants_chat_session_approval(self.history, self.id):
|
||||
return messages
|
||||
|
||||
# Keep the grant close to the latest user request so route-neutral
|
||||
# compaction/trimming preserves it. Copy the metadata instead of
|
||||
# mutating the durable transcript object.
|
||||
for index in range(len(messages) - 1, -1, -1):
|
||||
if messages[index].get("role") != "user":
|
||||
continue
|
||||
message = dict(messages[index])
|
||||
metadata = dict(message.get("metadata") or {})
|
||||
metadata[CHAT_SESSION_APPROVAL_CONTEXT_MARKER] = True
|
||||
message["metadata"] = metadata
|
||||
messages[index] = message
|
||||
break
|
||||
return messages
|
||||
|
||||
def get(self, key: str, default=None):
|
||||
"""Dict-like access for compatibility."""
|
||||
|
||||
+132
-37
@@ -14,8 +14,12 @@ import logging
|
||||
from datetime import datetime, timezone, timedelta
|
||||
from typing import Dict, Optional
|
||||
|
||||
from sqlalchemy import func
|
||||
|
||||
from .database import Session as DbSession, ChatMessage as DbChatMessage, Document as DbDocument, SessionLocal, utcnow_naive
|
||||
from .models import Session, ChatMessage
|
||||
from src.attachment_refs import persistable_message_content
|
||||
from src.upload_handler import reserve_message_upload_references
|
||||
|
||||
# Re-export singleton accessors from models for convenience
|
||||
from .models import set_session_manager_instance, get_session_manager_instance
|
||||
@@ -72,6 +76,7 @@ class SessionManager:
|
||||
def __init__(self, sessions_file: str = None):
|
||||
# sessions_file kept for backward compat, not used
|
||||
self.sessions: Dict[str, Session] = {}
|
||||
self.upload_handler = None
|
||||
self.load_sessions()
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -89,14 +94,28 @@ class SessionManager:
|
||||
try:
|
||||
db_sessions = db.query(DbSession).filter(
|
||||
DbSession.archived == False,
|
||||
DbSession.message_count > 0,
|
||||
DbSession.messages.any(),
|
||||
).order_by(DbSession.last_accessed.desc()).limit(100).all()
|
||||
|
||||
# message_count is derived metadata and can drift after interrupted
|
||||
# or legacy writes. Count only the bounded discovery set so startup
|
||||
# remains metadata-only while lazy hydration sees an authoritative
|
||||
# positive count for every discovered non-empty session.
|
||||
message_counts = {}
|
||||
if db_sessions:
|
||||
message_counts = dict(
|
||||
db.query(DbChatMessage.session_id, func.count(DbChatMessage.id))
|
||||
.filter(DbChatMessage.session_id.in_([row.id for row in db_sessions]))
|
||||
.group_by(DbChatMessage.session_id)
|
||||
.all()
|
||||
)
|
||||
|
||||
loaded_count = 0
|
||||
for db_session in db_sessions:
|
||||
try:
|
||||
session = self._db_to_session_meta(db_session)
|
||||
if session is not None:
|
||||
session.message_count = message_counts[db_session.id]
|
||||
self.sessions[db_session.id] = session
|
||||
loaded_count += 1
|
||||
except Exception as e:
|
||||
@@ -131,6 +150,12 @@ class SessionManager:
|
||||
history=[],
|
||||
owner=getattr(db_session, "owner", None),
|
||||
is_important=getattr(db_session, "is_important", False) or False,
|
||||
memory_extraction_enabled=getattr(db_session, "memory_extraction_enabled", True) is not False,
|
||||
skill_injection_enabled=getattr(db_session, "skill_injection_enabled", True) is not False,
|
||||
thinking_mode=getattr(db_session, "thinking_mode", "") or "off",
|
||||
temperature_override=getattr(db_session, "temperature_override", None),
|
||||
max_tokens_override=getattr(db_session, "max_tokens_override", None),
|
||||
cwd=getattr(db_session, "cwd", None) or None,
|
||||
)
|
||||
session.message_count = getattr(db_session, "message_count", 0) or 0
|
||||
return session
|
||||
@@ -189,9 +214,20 @@ class SessionManager:
|
||||
history=history,
|
||||
owner=getattr(db_session, 'owner', None),
|
||||
is_important=getattr(db_session, 'is_important', False) or False,
|
||||
memory_extraction_enabled=getattr(db_session, 'memory_extraction_enabled', True) is not False,
|
||||
skill_injection_enabled=getattr(db_session, 'skill_injection_enabled', True) is not False,
|
||||
thinking_mode=getattr(db_session, "thinking_mode", "") or "off",
|
||||
temperature_override=getattr(db_session, "temperature_override", None),
|
||||
max_tokens_override=getattr(db_session, "max_tokens_override", None),
|
||||
cwd=getattr(db_session, "cwd", None) or None,
|
||||
)
|
||||
|
||||
session.message_count = getattr(db_session, 'message_count', len(history))
|
||||
# The rows just loaded are the whole transcript, so they — not the
|
||||
# denormalized sessions.message_count column — are the truth for this
|
||||
# cached object. get_session's hydration gate compares against this
|
||||
# number; seeding it from a drifted column would ask for a reload that
|
||||
# can never close the gap.
|
||||
session.message_count = len(history)
|
||||
return session
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -230,17 +266,26 @@ class SessionManager:
|
||||
logger.warning("Dropping message for deleted session %s", session_id)
|
||||
return
|
||||
|
||||
missing_upload_id = reserve_message_upload_references(
|
||||
getattr(self, "upload_handler", None),
|
||||
getattr(db_session, "owner", None),
|
||||
message.content,
|
||||
message.metadata,
|
||||
)
|
||||
if missing_upload_id:
|
||||
raise ValueError(
|
||||
f"Referenced upload is no longer available: {missing_upload_id}"
|
||||
)
|
||||
|
||||
msg_id = str(uuid.uuid4())
|
||||
msg_time = datetime.utcnow()
|
||||
if message.metadata is None:
|
||||
message.metadata = {}
|
||||
message.metadata.setdefault('timestamp', _message_timestamp_iso(msg_time))
|
||||
# Multimodal content (image/audio attachments) is a list — serialize
|
||||
# to JSON so the Text column can store it. On reload, _db_to_session
|
||||
# detects the JSON-array prefix and parses it back.
|
||||
_content = message.content
|
||||
if isinstance(_content, list):
|
||||
_content = json.dumps(_content)
|
||||
# Multimodal content may contain provider data URLs for the live
|
||||
# model call. Persist only readable text plus attachment references
|
||||
# so chat_messages/FTS do not duplicate upload bytes.
|
||||
_content = persistable_message_content(message.content, message.metadata)
|
||||
db_message = DbChatMessage(
|
||||
id=msg_id,
|
||||
session_id=session_id,
|
||||
@@ -322,6 +367,28 @@ class SessionManager:
|
||||
session = self.get_session(session_id)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
db_session = db.query(DbSession).filter(DbSession.id == session_id).first()
|
||||
if db_session is None:
|
||||
logger.warning("Cannot replace history for missing session %s", session_id)
|
||||
return False
|
||||
|
||||
# Reserve every incoming attachment before removing any durable
|
||||
# message row. reserve_upload() shares the upload lifecycle lock
|
||||
# with cleanup, so an upload cannot be deleted between this
|
||||
# ownership check/access touch and the replacement transaction.
|
||||
# A failed reservation must leave the existing transcript intact.
|
||||
for message in messages:
|
||||
missing_upload_id = reserve_message_upload_references(
|
||||
getattr(self, "upload_handler", None),
|
||||
getattr(db_session, "owner", None),
|
||||
message.content,
|
||||
message.metadata,
|
||||
)
|
||||
if missing_upload_id:
|
||||
raise ValueError(
|
||||
f"Referenced upload is no longer available: {missing_upload_id}"
|
||||
)
|
||||
|
||||
db.query(DbChatMessage).filter(DbChatMessage.session_id == session_id).delete()
|
||||
now = datetime.now(timezone.utc)
|
||||
for i, message in enumerate(messages):
|
||||
@@ -330,15 +397,9 @@ class SessionManager:
|
||||
id=msg_id,
|
||||
session_id=session_id,
|
||||
role=message.role,
|
||||
# Multimodal content (image/audio attachments) is a list;
|
||||
# serialize to JSON so the Text column round-trips via
|
||||
# _parse_msg_content. Storing the raw list let SQLAlchemy
|
||||
# bind its single-quoted repr, which _parse_msg_content
|
||||
# cannot parse (it looks for double-quoted "type"), so the
|
||||
# attachment was destroyed on reload. Mirrors _persist_message.
|
||||
content=(json.dumps(message.content)
|
||||
if isinstance(message.content, list)
|
||||
else message.content),
|
||||
# Mirrors _persist_message: keep raw media bytes out of the
|
||||
# persisted transcript and search index.
|
||||
content=persistable_message_content(message.content, message.metadata),
|
||||
meta_data=json.dumps(message.metadata) if message.metadata else None,
|
||||
timestamp=now + timedelta(microseconds=i),
|
||||
)
|
||||
@@ -347,12 +408,10 @@ class SessionManager:
|
||||
message.metadata = {}
|
||||
message.metadata["_db_id"] = msg_id
|
||||
|
||||
db_session = db.query(DbSession).filter(DbSession.id == session_id).first()
|
||||
if db_session:
|
||||
db_session.message_count = len(messages)
|
||||
db_session.updated_at = now
|
||||
db_session.last_accessed = now
|
||||
db_session.last_message_at = now
|
||||
db_session.message_count = len(messages)
|
||||
db_session.updated_at = now
|
||||
db_session.last_accessed = now
|
||||
db_session.last_message_at = now
|
||||
|
||||
db.commit()
|
||||
session.history = list(messages)
|
||||
@@ -372,30 +431,50 @@ class SessionManager:
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def get_session(self, session_id: str) -> Session:
|
||||
"""Get a session by ID, loading from DB if needed.
|
||||
"""Get a session by ID, loading complete DB history when needed.
|
||||
|
||||
Sessions seeded by `load_sessions` start with empty history. The
|
||||
first read here hydrates them with the message rows.
|
||||
Sessions seeded by ``load_sessions`` start with empty history, and a
|
||||
cached session can also become partially stale. Refresh metadata first,
|
||||
then hydrate whenever the cached transcript is short of the stored rows.
|
||||
Model-send routes enter through this method before building context,
|
||||
while paginated display history reads SQLite directly.
|
||||
|
||||
The gate compares against ``sync_session_metadata``'s reconciled count
|
||||
(the real ``chat_messages`` total), never the denormalized column, so a
|
||||
hydrate always closes the gap and the next read is a cache hit.
|
||||
"""
|
||||
if session_id not in self.sessions:
|
||||
self._load_session_from_db(session_id)
|
||||
else:
|
||||
cached = self.sessions[session_id]
|
||||
# Lazy hydrate: metadata-only entries get their messages on first read.
|
||||
if not cached.history and getattr(cached, "message_count", 0) > 0:
|
||||
self._load_session_from_db(session_id)
|
||||
|
||||
# Keep model/endpoint metadata fresh. Endpoint deletion can clear the
|
||||
# DB row while a session object is still cached in RAM.
|
||||
# DB row while a session object is still cached in RAM. Refreshing first
|
||||
# also exposes the authoritative message count before completeness is
|
||||
# checked.
|
||||
self.sync_session_metadata(session_id)
|
||||
|
||||
cached = self.sessions[session_id]
|
||||
cached_count = len(cached.history or [])
|
||||
stored_count = int(getattr(cached, "message_count", 0) or 0)
|
||||
if cached_count < stored_count:
|
||||
self._load_session_from_db(session_id)
|
||||
|
||||
# Update last_accessed
|
||||
self._touch_session(session_id)
|
||||
|
||||
return self.sessions[session_id]
|
||||
|
||||
def sync_session_metadata(self, session_id: str) -> bool:
|
||||
"""Refresh non-message session fields from the DB into the cached object."""
|
||||
"""Refresh non-message session fields from the DB into the cached object.
|
||||
|
||||
``message_count`` is reconciled against the real ``chat_messages`` rows
|
||||
rather than copied from the denormalized ``sessions.message_count``
|
||||
column. That column drifts in normal operation — ``_persist_message``
|
||||
swallows a failed insert but ``add_message`` has already appended in
|
||||
memory, so the next successful persist writes rows+1, and a persist for
|
||||
an uncached session writes 0. Hydration keys off this number: a
|
||||
drifted-high column would reload the whole transcript on every warm
|
||||
read, and a drifted-low one would leave the model a truncated one.
|
||||
"""
|
||||
session = self.sessions.get(session_id)
|
||||
if session is None:
|
||||
return False
|
||||
@@ -418,7 +497,12 @@ class SessionManager:
|
||||
session.archived = db_session.archived
|
||||
session.owner = getattr(db_session, "owner", None)
|
||||
session.is_important = getattr(db_session, "is_important", False) or False
|
||||
session.message_count = getattr(db_session, "message_count", session.message_count) or 0
|
||||
session.cwd = getattr(db_session, "cwd", None) or None
|
||||
session.message_count = (
|
||||
db.query(DbChatMessage)
|
||||
.filter(DbChatMessage.session_id == session_id)
|
||||
.count()
|
||||
)
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error(f"Error syncing session metadata {session_id}: {e}")
|
||||
@@ -474,9 +558,12 @@ class SessionManager:
|
||||
endpoint_url: str,
|
||||
model: str,
|
||||
rag: bool = False,
|
||||
owner: str = None
|
||||
owner: str = None,
|
||||
cwd: str = None,
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
) -> Session:
|
||||
"""Create a new session and save to database."""
|
||||
session_headers = dict(headers or {})
|
||||
db = SessionLocal()
|
||||
try:
|
||||
db_session = DbSession(
|
||||
@@ -485,8 +572,9 @@ class SessionManager:
|
||||
endpoint_url=endpoint_url,
|
||||
model=model,
|
||||
rag=rag,
|
||||
headers={},
|
||||
headers=session_headers,
|
||||
owner=owner,
|
||||
cwd=cwd or None,
|
||||
created_at=datetime.now(timezone.utc),
|
||||
updated_at=datetime.now(timezone.utc)
|
||||
)
|
||||
@@ -499,8 +587,9 @@ class SessionManager:
|
||||
endpoint_url=endpoint_url,
|
||||
model=model,
|
||||
rag=rag,
|
||||
headers={},
|
||||
headers=session_headers,
|
||||
owner=owner,
|
||||
cwd=cwd or None,
|
||||
)
|
||||
|
||||
self.sessions[session_id] = session
|
||||
@@ -517,6 +606,12 @@ class SessionManager:
|
||||
"""Permanently delete a session and all its messages."""
|
||||
db = SessionLocal()
|
||||
try:
|
||||
try:
|
||||
from src.session_image_cleanup import cleanup_session_images
|
||||
cleanup_session_images(session_id, db=db)
|
||||
except Exception as e:
|
||||
logger.warning(f"Image cleanup failed while deleting session {session_id}: {e}")
|
||||
|
||||
# Detach documents so they survive as orphans in the library
|
||||
db.query(DbDocument).filter(DbDocument.session_id == session_id).update(
|
||||
{DbDocument.session_id: None}, synchronize_session=False
|
||||
|
||||
+29
-11
@@ -14,7 +14,7 @@ services:
|
||||
odysseus:
|
||||
build: .
|
||||
ports:
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7000}:7000"
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7011}:7000"
|
||||
volumes:
|
||||
- ${APP_DATA_DIR:-./data}:/app/data:z
|
||||
- ${APP_LOGS_DIR:-./logs}:/app/logs:z
|
||||
@@ -28,14 +28,6 @@ services:
|
||||
# land under /app/.local for the odysseus user. Persist them so a
|
||||
# container recreate does not silently remove installed serve engines.
|
||||
- ${APP_DATA_DIR:-./data}/local:/app/.local:z
|
||||
# Docker socket — lets Cookbook launch commands like
|
||||
# `docker exec ollama-rocm ollama show <tag>` reach the host's
|
||||
# Docker daemon (and sibling containers like ollama-rocm /
|
||||
# ollama-test). The in-container user needs to be in the
|
||||
# socket's owning group — see `group_add` below; the GID
|
||||
# there must match the host's `docker` group (defaults to 963
|
||||
# on Debian, 999 on Ubuntu — override via env if yours differs).
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
extra_hosts:
|
||||
# Lets the container reach local services on the Docker host, including
|
||||
# Ollama at http://host.docker.internal:11434.
|
||||
@@ -54,10 +46,11 @@ services:
|
||||
- DATABASE_URL=${DATABASE_URL:-sqlite:///./data/app.db}
|
||||
- AUTH_ENABLED=${AUTH_ENABLED:-true}
|
||||
- LOCALHOST_BYPASS=${LOCALHOST_BYPASS:-false}
|
||||
- COMPANION_BASE_URL=${COMPANION_BASE_URL:-}
|
||||
- ODYSSEUS_ADMIN_USER=${ODYSSEUS_ADMIN_USER:-admin}
|
||||
- ODYSSEUS_ADMIN_PASSWORD=${ODYSSEUS_ADMIN_PASSWORD:-}
|
||||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-http://localhost,http://127.0.0.1}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-false}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-}
|
||||
- EMBEDDING_URL=${EMBEDDING_URL:-}
|
||||
- EMBEDDING_MODEL=${EMBEDDING_MODEL:-}
|
||||
- EMBEDDING_API_KEY=${EMBEDDING_API_KEY:-}
|
||||
@@ -66,6 +59,11 @@ services:
|
||||
- CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24}
|
||||
- ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1}
|
||||
- ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1}
|
||||
- ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false}
|
||||
- ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1}
|
||||
- ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0}
|
||||
- ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0}
|
||||
- ODYSSEUS_MCP_EMAIL_OWNER=${ODYSSEUS_MCP_EMAIL_OWNER:-}
|
||||
- ODYSSEUS_SCRIPT_HOST=${ODYSSEUS_SCRIPT_HOST:-localhost}
|
||||
- ODYSSEUS_CHAT_UPLOAD_MAX_BYTES=${ODYSSEUS_CHAT_UPLOAD_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES=${ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES:-104857600}
|
||||
@@ -73,11 +71,27 @@ services:
|
||||
- ODYSSEUS_MEMORY_IMPORT_MAX_BYTES=${ODYSSEUS_MEMORY_IMPORT_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES=${ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES=${ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EDITOR_DRAFT_MAX_BYTES=${ODYSSEUS_EDITOR_DRAFT_MAX_BYTES:-268435456}
|
||||
- ODYSSEUS_STT_MAX_AUDIO_BYTES=${ODYSSEUS_STT_MAX_AUDIO_BYTES:-26214400}
|
||||
- ODYSSEUS_ICS_MAX_BYTES=${ODYSSEUS_ICS_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_TTS_CACHE_MAX_BYTES=${ODYSSEUS_TTS_CACHE_MAX_BYTES}
|
||||
# Host workspace translation is opt-in. Keep the public compose file
|
||||
# user-neutral; configure these in a local .env or use the host-workspace
|
||||
# overlay with ODYSSEUS_HOST_WORKSPACE_DIR.
|
||||
- ODYSSEUS_WORKSPACE_HOST_ROOT=${ODYSSEUS_WORKSPACE_HOST_ROOT:-}
|
||||
- ODYSSEUS_WORKSPACE_CONTAINER_ROOT=${ODYSSEUS_WORKSPACE_CONTAINER_ROOT:-/workspace}
|
||||
- ODYSSEUS_WORKSPACE_DEFAULT=${ODYSSEUS_WORKSPACE_DEFAULT:-}
|
||||
- DATA_BRAVE_API_KEY=${DATA_BRAVE_API_KEY:-}
|
||||
- GOOGLE_API_KEY=${GOOGLE_API_KEY:-}
|
||||
- GOOGLE_PSE_CX=${GOOGLE_PSE_CX:-}
|
||||
- GOOGLE_OAUTH_CLIENT_ID=${GOOGLE_OAUTH_CLIENT_ID:-}
|
||||
- GOOGLE_OAUTH_CLIENT_SECRET=${GOOGLE_OAUTH_CLIENT_SECRET:-}
|
||||
- GOOGLE_OAUTH_REDIRECT_URI=${GOOGLE_OAUTH_REDIRECT_URI:-}
|
||||
# Externally reachable origin for MCP OAuth callbacks. The container
|
||||
# always listens on 7000 and cannot see the host port map above, so
|
||||
# remote MCP OAuth needs this set whenever the browser reaches
|
||||
# Odysseus on anything other than http://localhost:7000.
|
||||
- OAUTH_REDIRECT_BASE_URL=${OAUTH_REDIRECT_BASE_URL:-}
|
||||
- TAVILY_API_KEY=${TAVILY_API_KEY:-}
|
||||
- SERPER_API_KEY=${SERPER_API_KEY:-}
|
||||
# PUID / PGID — the user/group the container drops to before
|
||||
@@ -101,7 +115,6 @@ services:
|
||||
- /dev/kfd
|
||||
- /dev/dri
|
||||
group_add:
|
||||
- "${DOCKER_GID:-963}"
|
||||
- video
|
||||
- ${RENDER_GID:-render}
|
||||
|
||||
@@ -134,12 +147,17 @@ services:
|
||||
fi
|
||||
sed "s|__SEARXNG_SECRET__|$$secret|g" /tmp/searxng-settings.yml.template > /etc/searxng/settings.yml
|
||||
fi
|
||||
# Advisory: a settings file the migration cannot parse or rewrite must
|
||||
# not be what stops searxng from booting. It explains itself on stderr
|
||||
# and we carry on, letting searxng report anything genuinely wrong.
|
||||
/usr/local/searxng/.venv/bin/python /tmp/migrate-searxng-settings.py /etc/searxng/settings.yml || true
|
||||
exec /usr/local/searxng/entrypoint.sh
|
||||
ports:
|
||||
- "127.0.0.1:8080:8080"
|
||||
volumes:
|
||||
- searxng-data:/etc/searxng
|
||||
- ./config/searxng/settings.yml:/tmp/searxng-settings.yml.template:ro,z
|
||||
- ./scripts/migrate_searxng_settings.py:/tmp/migrate-searxng-settings.py:ro,z
|
||||
environment:
|
||||
- SEARXNG_BASE_URL=http://localhost:8080/
|
||||
- SEARXNG_SECRET=${SEARXNG_SECRET:-}
|
||||
|
||||
@@ -13,7 +13,7 @@ services:
|
||||
odysseus:
|
||||
build: .
|
||||
ports:
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7000}:7000"
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7011}:7000"
|
||||
volumes:
|
||||
- ${APP_DATA_DIR:-./data}:/app/data:z
|
||||
- ${APP_LOGS_DIR:-./logs}:/app/logs:z
|
||||
@@ -27,16 +27,6 @@ services:
|
||||
# land under /app/.local for the odysseus user. Persist them so a
|
||||
# container recreate does not silently remove installed serve engines.
|
||||
- ${APP_DATA_DIR:-./data}/local:/app/.local:z
|
||||
# Docker socket — lets Cookbook launch commands like
|
||||
# `docker exec ollama-rocm ollama show <tag>` reach the host's
|
||||
# Docker daemon (and sibling containers like ollama-rocm /
|
||||
# ollama-test). The in-container user needs to be in the
|
||||
# socket's owning group — see `group_add` below; the GID
|
||||
# there must match the host's `docker` group (defaults to 963
|
||||
# on Debian, 999 on Ubuntu — override via env if yours differs).
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
group_add:
|
||||
- "${DOCKER_GID:-963}"
|
||||
extra_hosts:
|
||||
# Lets the container reach local services on the Docker host, including
|
||||
# Ollama at http://host.docker.internal:11434.
|
||||
@@ -55,10 +45,11 @@ services:
|
||||
- DATABASE_URL=${DATABASE_URL:-sqlite:///./data/app.db}
|
||||
- AUTH_ENABLED=${AUTH_ENABLED:-true}
|
||||
- LOCALHOST_BYPASS=${LOCALHOST_BYPASS:-false}
|
||||
- COMPANION_BASE_URL=${COMPANION_BASE_URL:-}
|
||||
- ODYSSEUS_ADMIN_USER=${ODYSSEUS_ADMIN_USER:-admin}
|
||||
- ODYSSEUS_ADMIN_PASSWORD=${ODYSSEUS_ADMIN_PASSWORD:-}
|
||||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-http://localhost,http://127.0.0.1}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-false}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-}
|
||||
- EMBEDDING_URL=${EMBEDDING_URL:-}
|
||||
- EMBEDDING_MODEL=${EMBEDDING_MODEL:-}
|
||||
- EMBEDDING_API_KEY=${EMBEDDING_API_KEY:-}
|
||||
@@ -67,6 +58,11 @@ services:
|
||||
- CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24}
|
||||
- ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1}
|
||||
- ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1}
|
||||
- ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false}
|
||||
- ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1}
|
||||
- ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0}
|
||||
- ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0}
|
||||
- ODYSSEUS_MCP_EMAIL_OWNER=${ODYSSEUS_MCP_EMAIL_OWNER:-}
|
||||
- ODYSSEUS_SCRIPT_HOST=${ODYSSEUS_SCRIPT_HOST:-localhost}
|
||||
- ODYSSEUS_CHAT_UPLOAD_MAX_BYTES=${ODYSSEUS_CHAT_UPLOAD_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES=${ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES:-104857600}
|
||||
@@ -74,11 +70,27 @@ services:
|
||||
- ODYSSEUS_MEMORY_IMPORT_MAX_BYTES=${ODYSSEUS_MEMORY_IMPORT_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES=${ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES=${ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EDITOR_DRAFT_MAX_BYTES=${ODYSSEUS_EDITOR_DRAFT_MAX_BYTES:-268435456}
|
||||
- ODYSSEUS_STT_MAX_AUDIO_BYTES=${ODYSSEUS_STT_MAX_AUDIO_BYTES:-26214400}
|
||||
- ODYSSEUS_ICS_MAX_BYTES=${ODYSSEUS_ICS_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_TTS_CACHE_MAX_BYTES=${ODYSSEUS_TTS_CACHE_MAX_BYTES}
|
||||
# Host workspace translation is opt-in. Keep the public compose file
|
||||
# user-neutral; configure these in a local .env or use the host-workspace
|
||||
# overlay with ODYSSEUS_HOST_WORKSPACE_DIR.
|
||||
- ODYSSEUS_WORKSPACE_HOST_ROOT=${ODYSSEUS_WORKSPACE_HOST_ROOT:-}
|
||||
- ODYSSEUS_WORKSPACE_CONTAINER_ROOT=${ODYSSEUS_WORKSPACE_CONTAINER_ROOT:-/workspace}
|
||||
- ODYSSEUS_WORKSPACE_DEFAULT=${ODYSSEUS_WORKSPACE_DEFAULT:-}
|
||||
- DATA_BRAVE_API_KEY=${DATA_BRAVE_API_KEY:-}
|
||||
- GOOGLE_API_KEY=${GOOGLE_API_KEY:-}
|
||||
- GOOGLE_PSE_CX=${GOOGLE_PSE_CX:-}
|
||||
- GOOGLE_OAUTH_CLIENT_ID=${GOOGLE_OAUTH_CLIENT_ID:-}
|
||||
- GOOGLE_OAUTH_CLIENT_SECRET=${GOOGLE_OAUTH_CLIENT_SECRET:-}
|
||||
- GOOGLE_OAUTH_REDIRECT_URI=${GOOGLE_OAUTH_REDIRECT_URI:-}
|
||||
# Externally reachable origin for MCP OAuth callbacks. The container
|
||||
# always listens on 7000 and cannot see the host port map above, so
|
||||
# remote MCP OAuth needs this set whenever the browser reaches
|
||||
# Odysseus on anything other than http://localhost:7000.
|
||||
- OAUTH_REDIRECT_BASE_URL=${OAUTH_REDIRECT_BASE_URL:-}
|
||||
- TAVILY_API_KEY=${TAVILY_API_KEY:-}
|
||||
- SERPER_API_KEY=${SERPER_API_KEY:-}
|
||||
# PUID / PGID — the user/group the container drops to before
|
||||
@@ -138,12 +150,17 @@ services:
|
||||
fi
|
||||
sed "s|__SEARXNG_SECRET__|$$secret|g" /tmp/searxng-settings.yml.template > /etc/searxng/settings.yml
|
||||
fi
|
||||
# Advisory: a settings file the migration cannot parse or rewrite must
|
||||
# not be what stops searxng from booting. It explains itself on stderr
|
||||
# and we carry on, letting searxng report anything genuinely wrong.
|
||||
/usr/local/searxng/.venv/bin/python /tmp/migrate-searxng-settings.py /etc/searxng/settings.yml || true
|
||||
exec /usr/local/searxng/entrypoint.sh
|
||||
ports:
|
||||
- "127.0.0.1:8080:8080"
|
||||
volumes:
|
||||
- searxng-data:/etc/searxng
|
||||
- ./config/searxng/settings.yml:/tmp/searxng-settings.yml.template:ro,z
|
||||
- ./scripts/migrate_searxng_settings.py:/tmp/migrate-searxng-settings.py:ro,z
|
||||
environment:
|
||||
- SEARXNG_BASE_URL=http://localhost:8080/
|
||||
- SEARXNG_SECRET=${SEARXNG_SECRET:-}
|
||||
|
||||
+29
-12
@@ -2,7 +2,7 @@ services:
|
||||
odysseus:
|
||||
build: .
|
||||
ports:
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7000}:7000"
|
||||
- "${APP_BIND:-127.0.0.1}:${APP_PORT:-7011}:7000"
|
||||
volumes:
|
||||
- ${APP_DATA_DIR:-./data}:/app/data:z
|
||||
- ${APP_LOGS_DIR:-./logs}:/app/logs:z
|
||||
@@ -16,16 +16,6 @@ services:
|
||||
# land under /app/.local for the odysseus user. Persist them so a
|
||||
# container recreate does not silently remove installed serve engines.
|
||||
- ${APP_DATA_DIR:-./data}/local:/app/.local:z
|
||||
# Docker socket — lets Cookbook launch commands like
|
||||
# `docker exec ollama-rocm ollama show <tag>` reach the host's
|
||||
# Docker daemon (and sibling containers like ollama-rocm /
|
||||
# ollama-test). The in-container user needs to be in the
|
||||
# socket's owning group — see `group_add` below; the GID
|
||||
# there must match the host's `docker` group (defaults to 963
|
||||
# on Debian, 999 on Ubuntu — override via env if yours differs).
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
group_add:
|
||||
- "${DOCKER_GID:-963}"
|
||||
extra_hosts:
|
||||
# Lets the container reach local services on the Docker host, including
|
||||
# Ollama at http://host.docker.internal:11434.
|
||||
@@ -44,10 +34,11 @@ services:
|
||||
- DATABASE_URL=${DATABASE_URL:-sqlite:///./data/app.db}
|
||||
- AUTH_ENABLED=${AUTH_ENABLED:-true}
|
||||
- LOCALHOST_BYPASS=${LOCALHOST_BYPASS:-false}
|
||||
- COMPANION_BASE_URL=${COMPANION_BASE_URL:-}
|
||||
- ODYSSEUS_ADMIN_USER=${ODYSSEUS_ADMIN_USER:-admin}
|
||||
- ODYSSEUS_ADMIN_PASSWORD=${ODYSSEUS_ADMIN_PASSWORD:-}
|
||||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-http://localhost,http://127.0.0.1}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-false}
|
||||
- SECURE_COOKIES=${SECURE_COOKIES:-}
|
||||
- EMBEDDING_URL=${EMBEDDING_URL:-}
|
||||
- EMBEDDING_MODEL=${EMBEDDING_MODEL:-}
|
||||
- EMBEDDING_API_KEY=${EMBEDDING_API_KEY:-}
|
||||
@@ -56,6 +47,11 @@ services:
|
||||
- CLEANUP_INTERVAL_HOURS=${CLEANUP_INTERVAL_HOURS:-24}
|
||||
- ODYSSEUS_INPROCESS_POLLERS=${ODYSSEUS_INPROCESS_POLLERS:-1}
|
||||
- ODYSSEUS_INPROCESS_TASKS=${ODYSSEUS_INPROCESS_TASKS:-1}
|
||||
- ODYSSEUS_UNATTENDED_MODE=${ODYSSEUS_UNATTENDED_MODE:-false}
|
||||
- ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS=${ODYSSEUS_QWEN_NATIVE_COMPACT_BUILTINS:-1}
|
||||
- ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT=${ODYSSEUS_QWEN_SUPPRESS_LOCAL_CONTEXT:-0}
|
||||
- ODYSSEUS_CAPTURE_MODEL_REQUESTS=${ODYSSEUS_CAPTURE_MODEL_REQUESTS:-0}
|
||||
- ODYSSEUS_MCP_EMAIL_OWNER=${ODYSSEUS_MCP_EMAIL_OWNER:-}
|
||||
- ODYSSEUS_SCRIPT_HOST=${ODYSSEUS_SCRIPT_HOST:-localhost}
|
||||
- ODYSSEUS_CHAT_UPLOAD_MAX_BYTES=${ODYSSEUS_CHAT_UPLOAD_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES=${ODYSSEUS_GALLERY_UPLOAD_MAX_BYTES:-104857600}
|
||||
@@ -63,11 +59,27 @@ services:
|
||||
- ODYSSEUS_MEMORY_IMPORT_MAX_BYTES=${ODYSSEUS_MEMORY_IMPORT_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES=${ODYSSEUS_PERSONAL_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES=${ODYSSEUS_EMAIL_COMPOSE_UPLOAD_MAX_BYTES:-26214400}
|
||||
- ODYSSEUS_EDITOR_DRAFT_MAX_BYTES=${ODYSSEUS_EDITOR_DRAFT_MAX_BYTES:-268435456}
|
||||
- ODYSSEUS_STT_MAX_AUDIO_BYTES=${ODYSSEUS_STT_MAX_AUDIO_BYTES:-26214400}
|
||||
- ODYSSEUS_ICS_MAX_BYTES=${ODYSSEUS_ICS_MAX_BYTES:-10485760}
|
||||
- ODYSSEUS_TTS_CACHE_MAX_BYTES=${ODYSSEUS_TTS_CACHE_MAX_BYTES}
|
||||
# Host workspace translation is opt-in. Keep the public compose file
|
||||
# user-neutral; configure these in a local .env or use the host-workspace
|
||||
# overlay with ODYSSEUS_HOST_WORKSPACE_DIR.
|
||||
- ODYSSEUS_WORKSPACE_HOST_ROOT=${ODYSSEUS_WORKSPACE_HOST_ROOT:-}
|
||||
- ODYSSEUS_WORKSPACE_CONTAINER_ROOT=${ODYSSEUS_WORKSPACE_CONTAINER_ROOT:-/workspace}
|
||||
- ODYSSEUS_WORKSPACE_DEFAULT=${ODYSSEUS_WORKSPACE_DEFAULT:-}
|
||||
- DATA_BRAVE_API_KEY=${DATA_BRAVE_API_KEY:-}
|
||||
- GOOGLE_API_KEY=${GOOGLE_API_KEY:-}
|
||||
- GOOGLE_PSE_CX=${GOOGLE_PSE_CX:-}
|
||||
- GOOGLE_OAUTH_CLIENT_ID=${GOOGLE_OAUTH_CLIENT_ID:-}
|
||||
- GOOGLE_OAUTH_CLIENT_SECRET=${GOOGLE_OAUTH_CLIENT_SECRET:-}
|
||||
- GOOGLE_OAUTH_REDIRECT_URI=${GOOGLE_OAUTH_REDIRECT_URI:-}
|
||||
# Externally reachable origin for MCP OAuth callbacks. The container
|
||||
# always listens on 7000 and cannot see the host port map above, so
|
||||
# remote MCP OAuth needs this set whenever the browser reaches
|
||||
# Odysseus on anything other than http://localhost:7000.
|
||||
- OAUTH_REDIRECT_BASE_URL=${OAUTH_REDIRECT_BASE_URL:-}
|
||||
- TAVILY_API_KEY=${TAVILY_API_KEY:-}
|
||||
- SERPER_API_KEY=${SERPER_API_KEY:-}
|
||||
# PUID / PGID — the user/group the container drops to before
|
||||
@@ -116,12 +128,17 @@ services:
|
||||
fi
|
||||
sed "s|__SEARXNG_SECRET__|$$secret|g" /tmp/searxng-settings.yml.template > /etc/searxng/settings.yml
|
||||
fi
|
||||
# Advisory: a settings file the migration cannot parse or rewrite must
|
||||
# not be what stops searxng from booting. It explains itself on stderr
|
||||
# and we carry on, letting searxng report anything genuinely wrong.
|
||||
/usr/local/searxng/.venv/bin/python /tmp/migrate-searxng-settings.py /etc/searxng/settings.yml || true
|
||||
exec /usr/local/searxng/entrypoint.sh
|
||||
ports:
|
||||
- "127.0.0.1:8080:8080"
|
||||
volumes:
|
||||
- searxng-data:/etc/searxng
|
||||
- ./config/searxng/settings.yml:/tmp/searxng-settings.yml.template:ro,z
|
||||
- ./scripts/migrate_searxng_settings.py:/tmp/migrate-searxng-settings.py:ro,z
|
||||
environment:
|
||||
- SEARXNG_BASE_URL=http://localhost:8080/
|
||||
- SEARXNG_SECRET=${SEARXNG_SECRET:-}
|
||||
|
||||
@@ -29,12 +29,12 @@ fi
|
||||
ODY_USER="$(getent passwd "$PUID" | cut -d: -f1)"
|
||||
[ -z "$ODY_USER" ] && ODY_USER=odysseus
|
||||
|
||||
# Docker-socket group plumbing. When /var/run/docker.sock is bind-mounted
|
||||
# (Cookbook uses docker exec to reach sibling containers), the socket is
|
||||
# owned by root:<host docker gid>. Add the app user to that group and later
|
||||
# call gosu by username so supplementary groups are retained.
|
||||
# Docker-socket group plumbing for the explicit host-Docker overlay. When
|
||||
# opted in, the socket is owned by root:<host docker gid>. Add the app user
|
||||
# to that group and later call gosu by username so supplementary groups are
|
||||
# retained.
|
||||
DOCKER_SOCK="${DOCKER_SOCK:-/var/run/docker.sock}"
|
||||
if [ -S "$DOCKER_SOCK" ]; then
|
||||
if [ "${ODYSSEUS_ENABLE_HOST_DOCKER:-}" = "true" ] && [ -S "$DOCKER_SOCK" ]; then
|
||||
SOCK_GID="$(stat -c '%g' "$DOCKER_SOCK" 2>/dev/null || echo '')"
|
||||
if [ -n "$SOCK_GID" ] && [ "$SOCK_GID" != "0" ]; then
|
||||
if ! getent group "$SOCK_GID" >/dev/null 2>&1; then
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
# High-trust host Docker access. Enable only when local Docker-daemon
|
||||
# management from Cookbook is required and you accept that raw socket access
|
||||
# grants broad control over the host Docker daemon.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-docker.yml
|
||||
# DOCKER_GID=<numeric host Docker group id>
|
||||
services:
|
||||
odysseus:
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
group_add: ["${DOCKER_GID:-963}"]
|
||||
environment:
|
||||
- ODYSSEUS_ENABLE_HOST_DOCKER=true
|
||||
@@ -0,0 +1,21 @@
|
||||
# High-trust host network access. Enable only when the Odysseus agent needs
|
||||
# host-native LAN/VPN/mDNS behavior that Docker bridge networking cannot
|
||||
# provide. Linux only; Docker Desktop does not provide equivalent host
|
||||
# networking semantics.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml:docker/host-network.yml
|
||||
# APP_PORT=7011
|
||||
services:
|
||||
odysseus:
|
||||
network_mode: host
|
||||
ports: !reset []
|
||||
environment:
|
||||
- APP_PORT=${APP_PORT:-7011}
|
||||
- APP_BIND=${APP_BIND:-0.0.0.0}
|
||||
- SEARXNG_INSTANCE=${ODYSSEUS_HOST_NETWORK_SEARXNG_INSTANCE:-http://127.0.0.1:8080}
|
||||
- CHROMADB_HOST=${ODYSSEUS_HOST_NETWORK_CHROMADB_HOST:-127.0.0.1}
|
||||
- CHROMADB_PORT=${ODYSSEUS_HOST_NETWORK_CHROMADB_PORT:-8100}
|
||||
- ODYSSEUS_CONTAINER_NETWORK_MODE=host
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- exec uvicorn app:app --host "$${APP_BIND:-0.0.0.0}" --port "$${APP_PORT:-7011}"
|
||||
@@ -0,0 +1,11 @@
|
||||
# High-trust host workspace access. Enable only when the Odysseus agent should
|
||||
# work on a host directory outside the container's normal /app/data sandbox.
|
||||
# COMPOSE_FILE=docker-compose.yml:docker/host-workspace.yml
|
||||
# ODYSSEUS_HOST_WORKSPACE_DIR=/absolute/host/path
|
||||
# ODYSSEUS_HOST_WORKSPACE_MOUNT=/host/workspace
|
||||
services:
|
||||
odysseus:
|
||||
volumes:
|
||||
- ${ODYSSEUS_HOST_WORKSPACE_DIR:?set ODYSSEUS_HOST_WORKSPACE_DIR}:${ODYSSEUS_HOST_WORKSPACE_MOUNT:-/host/workspace}:rw,z
|
||||
environment:
|
||||
- ODYSSEUS_HOST_WORKSPACE_MOUNT=${ODYSSEUS_HOST_WORKSPACE_MOUNT:-/host/workspace}
|
||||
@@ -0,0 +1,75 @@
|
||||
# Agent turn contract
|
||||
|
||||
Scope: product Agent turns on 7011. Environment-owned native/TUI bridges retain
|
||||
their existing execution contract. No model weights or training settings change.
|
||||
|
||||
## Boundaries
|
||||
|
||||
1. `src/turn_contract.py` classifies capabilities, including explicit compound
|
||||
requests and referential follow-ups. Classification is selection, not permission.
|
||||
2. `routes/chat_routes.py` resolves toggles, privileges, global/plan/incognito
|
||||
restrictions, fixture restrictions and available schema inventory before
|
||||
freezing the offered set. Web enabled alone does not select web tools.
|
||||
3. `TurnContract` checks `required <= offered <= executable`, stores immutable
|
||||
serialized schema copies, and records unavailable requirements. An unavailable
|
||||
request stops without inference or substitution; unknown actions ask for clarity.
|
||||
Exact account-discovery requests narrow selection to account metadata only;
|
||||
compounds retain their declared family scope. Media operations declare their
|
||||
existing tool dependencies rather than falling back to shell generation.
|
||||
4. The agent's prompt/schema route and fallback use that same logical scope.
|
||||
Native versus textual serialization remains model-specific. Answer-only phases
|
||||
can suppress tool calls without granting a different scope.
|
||||
Contract turns preserve the already-compacted conversation and tool-call/result
|
||||
IDs. The standalone specialist prompt's latest-message-only behavior is not used
|
||||
for these product turns. Prompt domains also come from the contract.
|
||||
Accepted in-scope calls retain their model-provided arguments and native IDs;
|
||||
the explicit-intent fallback must not overwrite them with the whole user turn.
|
||||
5. The context-bound dispatcher checks membership **and** existing runtime policy,
|
||||
owner restrictions and exact-action approvals. A contract is not authorization
|
||||
to bypass those gates. Contract work bypasses terminating legacy shortcuts.
|
||||
6. `_AgentRenderState` explicitly identifies streamed versus canonical output.
|
||||
Later synthesis transfers ownership with turn-scoped replacement. The frontend
|
||||
reconciles visible DOM, not just accumulated strings; tool evidence is retained.
|
||||
Ownership is included in saved metrics and `message_saved` events.
|
||||
History and resume honor replacement scope. Single-capability turns retain
|
||||
canonical output: an always-synthesize trial caused a live notes loop and was
|
||||
reverted. Compound turns cannot terminate after only one capability's result.
|
||||
|
||||
## Verification
|
||||
|
||||
Use the project's configured Python environment, not an unrelated system Python:
|
||||
|
||||
```sh
|
||||
/home/pewds/odysseus-cookbook-fresh/.venv/bin/python -m pytest -q \
|
||||
tests/test_turn_contract.py tests/test_turn_contract_integration.py \
|
||||
tests/test_agent_turn_contract_boundaries.py tests/test_turn_rendering_js.py \
|
||||
tests/test_contract_prompt_conversation.py tests/test_product_turn_contract_route.py \
|
||||
tests/test_contract_explicit_fallback.py \
|
||||
tests/test_history_resume_rendering_js.py \
|
||||
tests/test_chat_route_tool_policy.py tests/test_tool_policy.py \
|
||||
tests/test_frontend_module_version_parity.py
|
||||
node scripts/verify_agent_turn_contract.mjs --max-turns 80 --total-ms 900000
|
||||
```
|
||||
|
||||
The browser verifier uses `sft_alex_creator` and actual 7011 Agent controls. It
|
||||
captures request toggles, SSE contract/tool events, visible output and persisted
|
||||
history. Ten families have four initial/follow-up Web-toggle combinations.
|
||||
Blocked or unrun cases are not passes. Email requires verified fixture isolation;
|
||||
do not enable global fixture mode on the user's live service to make a test pass.
|
||||
|
||||
## Remaining limits
|
||||
|
||||
- Classification is deterministic and vocabulary-based, not a proof of semantic
|
||||
understanding. Add independent behavior examples for confirmed misses.
|
||||
- Schema registration and policy permission do not guarantee a remote provider
|
||||
stays healthy throughout a turn. Runtime failure must remain visible.
|
||||
- Separate tool/argument errors, tool-service failures, rendering failures and
|
||||
verifier defects in reports. Do not infer model accuracy from routing alone.
|
||||
- Canonical summaries can still ignore presentation constraints such as a
|
||||
requested item count. Do not count those as full functional passes. Forcing an
|
||||
extra model round is not a validated general repair for this deployed model.
|
||||
- Keep all imports of a local JS module on the same URL identity. Distinct query
|
||||
versions instantiate separate module state even when source files are identical.
|
||||
|
||||
Live baseline and current matrix results are in `reports/agent-turn-contract-*`.
|
||||
The implementation is not a claim that every family has passed live verification.
|
||||
@@ -0,0 +1,55 @@
|
||||
# Background research → originating chat
|
||||
|
||||
Chat `trigger_research` calls carry a **dispatcher-supplied** `origin_chat_id`.
|
||||
The research start route verifies chat ownership before registering a durable
|
||||
`background_tool_jobs` row and starting the existing research service. Panel
|
||||
jobs have no origin and never inject a chat reply.
|
||||
|
||||
- Chat default: **2 rounds**, 120-second *soft* research budget. Explicit
|
||||
deeper/Auto rounds regain the normal research time budget. Panel defaults
|
||||
remain unchanged. This is not a guaranteed two-minute wall-clock deadline.
|
||||
- A completion callback stores the report and sources. A startup worker also
|
||||
reconciles missed callbacks and research errors/restarts.
|
||||
- When the origin has no active foreground/detached run, its model summarizes
|
||||
the report with thinking off and no tools. An outer 75-second deadline also
|
||||
bounds model-slot waits. If synthesis is unavailable, deliver an honest
|
||||
notice plus the report link; preserve the evidence for follow-ups.
|
||||
- Message and delivery marker commit in one transaction with a deterministic
|
||||
message ID. Report context is stored in server message metadata and injected
|
||||
as untrusted evidence in regular and compact model history. Long excerpts
|
||||
are explicitly marked; the saved full research report remains accessible.
|
||||
- The browser polls owner-scoped `/api/research/chat-jobs/{chat_id}`, appending
|
||||
unseen message IDs only when that chat is current and not streaming. No
|
||||
transcript replacement or forced navigation. Reloaded history deduplicates.
|
||||
- Chat uses the existing agent-thread rail and expandable rows. The compact
|
||||
header shows status and a right-aligned BG task label with the shared whirlpool
|
||||
while running; expanding reveals topic, phase/round, source count and report
|
||||
link. Rows update in place, preserving expansion/focus while chat streams.
|
||||
Completed rows remain visible; zero-source runs show a warning, not success.
|
||||
Progress polling excludes reports and internal fields.
|
||||
|
||||
Other tools are **not automatically backgrounded**. The durable handoff can be
|
||||
reused, but each future producer needs explicit launch/result/permission wiring.
|
||||
|
||||
## Verification
|
||||
|
||||
```sh
|
||||
/home/pewds/odysseus-cookbook-fresh/.venv/bin/pytest -q tests/test_background_tool_jobs.py tests/test_research_chat_runtime.py
|
||||
node --test tests/backgroundToolJobs.test.mjs
|
||||
node scripts/verify_background_delivery_isolation.mjs
|
||||
node scripts/verify_background_research_cards.mjs
|
||||
node scripts/verify_background_research_chat.mjs
|
||||
```
|
||||
|
||||
The last script uses disposable `sft_alex_creator` chats and real research/model
|
||||
calls, then removes only its own reports/chats. Do not use real-user mutations.
|
||||
It checks two-round launch, continued chat, automatic arrival, no transcript
|
||||
rebuild/duplicates, reload, and a follow-up. Inspect retained report excerpts
|
||||
and generated summary when it fails; do not equate job launch with good research.
|
||||
|
||||
Initial live runs verified delivery/navigation/follow-ups but exposed a summary
|
||||
attempt-count bug (fixed: helper requires **1 attempt**, not `max_retries=0`).
|
||||
A later full run was interrupted by an inference endpoint outage. The corrected
|
||||
summary path separately passed a real-model evidence/limitations/citation probe.
|
||||
All targeted Python tests passed (441); real DOM isolation checks passed. A clean
|
||||
full live run with useful retrieved evidence remains to be recorded.
|
||||
@@ -0,0 +1,99 @@
|
||||
# No-RAG clean loop: first diagnostic
|
||||
|
||||
## Setup
|
||||
|
||||
No live UI, service configuration, or weights changed. The standalone loop sends
|
||||
conversation history, native assistant calls and matching tool results directly
|
||||
to the served pre-Heretic model. It never rewrites queries, invents calls, swaps
|
||||
families, or strips output. Invalid calls return errors. Six executions per turn
|
||||
and seven model rounds bound the test.
|
||||
|
||||
Both arms use temperature 0, thinking disabled, 768 output tokens, and the
|
||||
original tool-work evaluator's `tools_for_mode(..., 'compact_contract_v3')`.
|
||||
This matters: the app's plain compact scrubber deletes descriptions, whereas v3
|
||||
retains empirically tested micro-hints. Previous plain-compact tests were not
|
||||
exact reproductions of the passing benchmark setup.
|
||||
|
||||
The 76 tools come from the current app's ten-family inventory, transformed by
|
||||
the original v3 builder. This is not a byte-identical frozen 99-tool benchmark
|
||||
inventory or proof of training-data identity. The report records schema and
|
||||
builder hashes. No schemas are invented for this experiment.
|
||||
|
||||
- **Stable:** same compact inventory on every turn, irrespective of spelling.
|
||||
- **Routed:** same loop, but existing `requested_capabilities` chooses inventory
|
||||
each turn. This isolates that selector; it is not the complete production RAG
|
||||
or Agent UI path. Other production normalizers are absent in both arms.
|
||||
- Private records are synthetic. No real private dispatcher is imported.
|
||||
Only fixture reads and optional public SearXNG calls execute. Other operations
|
||||
return explicit errors, so this does not validate their functionality.
|
||||
- Live search sends the exact model query to local SearXNG Bing/Yep, bypassing
|
||||
app query rewriting/filtering. Source results may vary between arms.
|
||||
|
||||
## Observations, not a blind score
|
||||
|
||||
| Case | Stable compact inventory | Selector arm |
|
||||
|---|---|---|
|
||||
| `whats the current stock mraket` | Selected `web_search`, query `current stock market` | Offered zero tools; declined live lookup |
|
||||
| Exact seeded failed exchange, then `can you look up` | Searched with corrected query | Also searched with corrected query |
|
||||
| Summarize search, explicitly no tools | Answered without tools or permission failure | Same |
|
||||
| Calendar → email → calendar | Recalled second event at 14:30 | Same |
|
||||
| Notes → second note → what does it say | Correct `view` ID and content | Same after fixture correction |
|
||||
| Deliberately irrelevant search result | Did not automatically retry | Did not automatically retry |
|
||||
| User asks for a better source | Refined and executed another search | Proposed search was not offered and was rejected |
|
||||
| Web-disabled lookup | Attempted network access via bash; sandbox rejected it | Invented unsupported current market news without tools |
|
||||
|
||||
The initial stable stock answer listed sources, not current index values. It
|
||||
does not establish that the market question was fully answered. Its subsequent
|
||||
`can you look up` elicited clarification after it had already searched. The
|
||||
separate seeded replay removes that differing-history confound.
|
||||
|
||||
The first notes fixture incorrectly accepted `get/read`, not the real `view`
|
||||
action. Both models selected the correct action, but the fixture rejected it.
|
||||
Those six original turns are invalid for execution comparison. A corrected
|
||||
six-turn rerun succeeded in both arms; the failed evidence is retained.
|
||||
|
||||
Web-off results are a release blocker: removing named web tools alone does not
|
||||
enforce network denial across general-purpose tools. The fixture prevented real
|
||||
execution, but any UI integration must use the real cross-tool permissions and
|
||||
clearly communicate unavailable capabilities. Neither arm is ready for a live
|
||||
switch. Source recovery and grounded completion also remain weak.
|
||||
|
||||
## What this changes
|
||||
|
||||
There is direct evidence that the selector can withhold needed tools, and that
|
||||
the model can repair the misspelled query itself when offered the tool. Clean
|
||||
history also supports the tested topic switches without synthetic substitutions.
|
||||
This supports continuing the clean-path experiment, not retraining or declaring
|
||||
the UI fixed. Full inventory is slower in these requests; overlapping runs and
|
||||
different source content prevent a controlled latency conclusion.
|
||||
|
||||
Next: integrate the clean loop behind a test-only UI profile with real permission
|
||||
enforcement and one renderer, preserving the v3 contract. Test live read-only
|
||||
follow-ups and explicit Web-off behavior before any rollout. Separately compare
|
||||
a generic evidence-check/retry instruction on the weak-result fixture; do not
|
||||
manufacture a retry query in the harness.
|
||||
|
||||
## Reproduce
|
||||
|
||||
Eight boundary tests pass:
|
||||
|
||||
```sh
|
||||
/home/pewds/odysseus-cookbook-fresh/.venv/bin/pytest -q tests/test_clean_tool_loop.py
|
||||
```
|
||||
|
||||
Run with a fresh report filename (existing evidence is never overwritten):
|
||||
|
||||
```sh
|
||||
/home/pewds/odysseus-cookbook-fresh/.venv/bin/python scripts/test_clean_tool_loop.py --live-search --report reports/clean-loop-v3-new-run.json
|
||||
```
|
||||
|
||||
Evidence:
|
||||
|
||||
- `reports/clean-loop-v3-20260909.json`: original 24 turns; notes fixture caveat above.
|
||||
- `reports/clean-loop-v3-stock-seeded-20260909.json`: four matched seeded follow-up turns.
|
||||
- `reports/clean-loop-v3-notes-fixture-corrected-20260909.json`: corrected six notes turns.
|
||||
|
||||
Each report retains model requests, responses, offered inventory and execution
|
||||
results. The `completed` status means the request loop finished, **not** that
|
||||
the answer passed functional evaluation. These are synthetic/public traces, not
|
||||
private user conversations. This test does not measure UI rendering or streaming.
|
||||
@@ -0,0 +1,220 @@
|
||||
# Tools v3 — No-RAG preview
|
||||
|
||||
Select this endpoint in the 7011 model picker, with model
|
||||
`odysseus-qwen3.5-tools-pre-heretic`. This endpoint owns its complete tool loop
|
||||
and enters Agent mode server-side on every turn, including ambiguous follow-ups;
|
||||
it does not depend on the legacy per-message intent classifier. Start a new chat
|
||||
for an uncontaminated comparison. Enable Web for searches. Clean routing is
|
||||
owned by the exact model identity, so both the normal `preheret` endpoint and
|
||||
the `cleanv3` alias use this runtime. Every other model remains on legacy RAG.
|
||||
|
||||
Endpoint ID: `cleanv3`. Its base URL uses the same inference server's Tailscale
|
||||
DNS name, `http://odysseus.tailb895f4.ts.net:18182/v1`, to distinguish it from
|
||||
the original IP-address route when existing chats omit endpoint IDs.
|
||||
|
||||
## Implementation
|
||||
|
||||
- `src/clean_agent_preview.py` is a separate streamed native-tool loop, entered
|
||||
before legacy routing and substitutions. It uses real authenticated tool
|
||||
dispatch, the tool-work `compact_contract_v5` builder, temperature 0,
|
||||
and thinking disabled. No weights change or inference server was started.
|
||||
- The offered tool inventory is stable except for permissions/toggles. Safe,
|
||||
explicit personal creates/updates are enabled for notes, tasks, calendar,
|
||||
memory, skills and documents. Destructive operations, shell/code, outbound
|
||||
email, browser interaction, deployment/admin changes and unrelated-family
|
||||
write substitution remain blocked. No tool or argument substitution is
|
||||
applied by the loop.
|
||||
- Native calls and matching results persist in `clean_v3_turn` metadata so
|
||||
follow-ups use actual evidence. History retains at most eight complete turns,
|
||||
trimming oldest whole turns for size; individual outputs cap at 8000 chars.
|
||||
- Real search still uses the existing search backend and its provider handling;
|
||||
this does not claim that provider quality or every backend transform is fixed.
|
||||
- All routing, privileges and default settings outside this exact Odysseus model
|
||||
remain unchanged. The loop has six execution/eight-round limits.
|
||||
- Write completion is evidence-bound: affirmative success text is replaced
|
||||
unless a private-write tool succeeded during the turn. Proposed call batches
|
||||
are policy-preflighted atomically, so a batch containing a blocked operation
|
||||
cannot partially execute before denial.
|
||||
|
||||
## Verification
|
||||
|
||||
399 focused Python tests passed after route integration. Browser runs r1/r2
|
||||
accidentally exercised the old loop and are not preview evidence. The runner
|
||||
now explicitly asserts `selection_mode=clean_compact_v3_preview`.
|
||||
|
||||
`reports/clean-v3-live-ui-r3-20260909.json` confirms the preview route, real notes
|
||||
execution, correct repetition from history, successful search and no-tool
|
||||
summary, plus visible incremental growth. Its notes assertions were for the
|
||||
old routed contract: they prohibited offering web tools even with Web enabled,
|
||||
and required another notes call for a verbatim repeat. The updated preview
|
||||
checks permit stable offers and accept an exact match to the preceding saved
|
||||
answer without re-execution; execution permissions are still asserted.
|
||||
|
||||
`reports/clean-v3-live-ui-r4-20260909.json` is the corrected four-turn check,
|
||||
including notes with Web off and search with Web on: **4/4 passed**, with the
|
||||
preview selection mode explicitly confirmed on every turn.
|
||||
These are UI smoke tests, not all-family or factual-answer benchmark scores.
|
||||
|
||||
## Disable
|
||||
|
||||
Disabling only endpoint `cleanv3` removes the duplicate picker alias; it does
|
||||
not disable this model-owned runtime. To roll back the runtime, revert the exact
|
||||
model route in `routes/chat_routes.py`. Do not delete weights, adapters, or user
|
||||
chats. The v3 schema builder dependency is
|
||||
`/home/pewds/odysseus-tool-work/scripts/eval_alltools_unseen_compare.py` and its
|
||||
schema-dropout helper; preserve those with this deployment.
|
||||
|
||||
## Expanded UI checks — 2026-09-09
|
||||
|
||||
24 additional turns completed through the preview: 23 automated passes and one
|
||||
checker false alarm. The Cookbook follow-up correctly shortened the previous
|
||||
six-server result to the first three requested names without another call. The
|
||||
checker required either a fresh call or a verbatim repeat; manual inspection
|
||||
confirmed the requested subset. Raw failure evidence is retained, not rescored.
|
||||
|
||||
Covered notes/misspellings/second-note selection, calendar/second-event time,
|
||||
tasks, documents, memory, skills, Cookbook listing, misspelled search, and Web
|
||||
toggle changes. Cross-family flows passed: Germany news → “whats my notes”,
|
||||
notes → “seach current stock mraket news”, and calendar → “now show my noes”.
|
||||
The model chose `current stock market news` itself. Every completed turn's
|
||||
audit confirmed the preview mode. Search source factual accuracy is not graded
|
||||
by this suite, and successful reads do not establish mutation coverage.
|
||||
|
||||
Email was separately attempted but the test guard stopped it because the
|
||||
stable offered inventory exceeded its metadata-only verified scope. Email
|
||||
therefore remains unverified in this expanded run; the guard was not weakened.
|
||||
No production code, service settings or weights changed during these tests.
|
||||
|
||||
Evidence under `reports/`:
|
||||
|
||||
- `clean-v3-broader-ui-20260909.json`: 16 turns, 15 automatic passes, Cookbook caveat.
|
||||
- `clean-v3-topic-switch-ui-20260909.json`: 6/6 passed.
|
||||
- `clean-v3-second-note-ui-20260909.json`: 2/2 passed.
|
||||
- `clean-v3-email-notes-ui-20260909.json`: blocked email attempt; notes not run in that file.
|
||||
|
||||
## Picker route fix
|
||||
|
||||
The previous tests selected sessions through the API, missing a real picker
|
||||
bug: local entries were deduplicated by model ID, hiding alternative endpoints
|
||||
with the same weights. The picker now uses endpoint+model identity for local
|
||||
routes too, displays the endpoint name, and scopes its last-picked send override
|
||||
to the current chat. `/api/sessions` returns owner-filtered endpoint identity
|
||||
for unambiguous saved URLs, so reload labels do not depend on loading the model
|
||||
catalog. Ambiguous identical URLs are not guessed.
|
||||
|
||||
The user-authorized chat `ec0683a2-015f-41d7-aa1f-34135c9640cb` was switched to
|
||||
`cleanv3` using the authenticated session PATCH API; no messages were inserted
|
||||
and no tool actions ran in that chat. Defaults and other chats were unchanged.
|
||||
|
||||
The runner's `--picker-route true` starts on the original route, clicks the
|
||||
preview in the real picker, sends a greeting, reloads the chat permalink, then
|
||||
asks for notes. Early picker/reload reports are incomplete, not passes: their
|
||||
label check exposed the unloaded-catalog issue. Focused route/picker/history
|
||||
tests: 16 passed.
|
||||
|
||||
Final picker test: `reports/clean-v3-picker-reload-r5-20260909.json`, **2/2
|
||||
passed**. Real picker click, greeting, permalink reload, and notes follow-up
|
||||
all confirmed the preview route. The label survived reload. R4 retained a
|
||||
history/DOM mismatch from sending before restored history was ready; the final
|
||||
driver explicitly waits for the saved first answer to render before sending.
|
||||
This does not claim a general fix for sending during unfinished history loading.
|
||||
|
||||
## Native image/VL status
|
||||
|
||||
The inference launcher previously set `--limit-mm-per-prompt` to zero images,
|
||||
so vLLM rejected attachments before the model saw them. The durable Odysseus
|
||||
launcher now permits up to three images per prompt; video remains disabled.
|
||||
|
||||
`reports/clean-v3-vl-live-r4-20260909.json` proves the real 7011 attachment
|
||||
path, clean compact route, object/color/spatial recognition, permalink reload,
|
||||
and ambiguous image follow-up. Those checks pass. Exact OCR of the deterministic
|
||||
`ODYSSEUS 42` heading fails in both the untouched Qwen 3.5 9B base and the
|
||||
fine-tune, so it remains a base/runtime capability limitation rather than a
|
||||
fine-tune regression or harness failure.
|
||||
|
||||
The same native path also passes JPEG and lossless WebP transport, object
|
||||
recognition, reload, and follow-up grounding. Evidence:
|
||||
`reports/clean-v3-vl-jpeg-r1-20260909.json` and
|
||||
`reports/clean-v3-vl-webp-r1-20260909.json`. Both remain `partial` only because
|
||||
the shared OCR check fails.
|
||||
|
||||
## Reversible write check
|
||||
|
||||
`scripts/verify_clean_v3_write.mjs` runs against only `sft_alex_creator`. It
|
||||
creates one UUID-named note through the real 7011 UI, verifies that exact row,
|
||||
requests a destructive bulk deletion, verifies the row still exists, and then
|
||||
deletes only its own test row through the authenticated API. The cleanup is
|
||||
verified by a 404 lookup.
|
||||
|
||||
Final evidence: `reports/clean-v3-write-ui-r8-20260909.json`, **passed**. Both
|
||||
turns reported `selection_mode=clean_compact_v3_preview`; creation executed via
|
||||
`manage_notes(action=add)`, the destructive action did not execute, and the
|
||||
canonical response was “No changes were made.” The earlier r3/r5 files are
|
||||
startup/placement failures, while r4/r6/r7 retained genuine intermediate
|
||||
harness and verifier failures; none should be interpreted as passes.
|
||||
|
||||
## Stateful, search, and email checks
|
||||
|
||||
The reversible stateful runner passes all six mutation families in one run:
|
||||
calendar, notes, tasks, documents, memory, and skills (**6/6**). Each flow
|
||||
creates a UUID-only artifact through the real Agent UI, verifies it by
|
||||
owner-scoped API, applies a noun-free correction, verifies persistence, and
|
||||
removes only that artifact. A direct database audit found zero active synthetic
|
||||
calendar, note, task, or document rows afterward.
|
||||
|
||||
The document failure was harness-owned. Compact description dropout left a
|
||||
vague free-form `command` field, error envelopes defaulted to exit code 0, and
|
||||
the clean loop dropped the active document ID. Compact v5 now exposes only
|
||||
required structured `edits`, reports errors truthfully, and executes against
|
||||
the request's explicit active document. Fresh document and combined stateful
|
||||
runs pass.
|
||||
|
||||
Search Web-toggle combinations `00`, `01`, `10`, and `11` pass **8/8** across
|
||||
two turns. A web question can no longer silently enable Bash because it says
|
||||
“official source”, and an unavailable Web capability exposes no unrelated
|
||||
fallback family. The quality suite passes **3/3**: evidence reuse without a
|
||||
second call, explicit official-page inspection with `web_fetch`, correction of
|
||||
“stock mraket” in actual search arguments, and a truthful unsupported result
|
||||
for a synthetic company.
|
||||
|
||||
Production-path email reads pass **3/3** through the running email MCP: account
|
||||
list, latest inbox list, and referential read of the first result. The report
|
||||
retains no account names, addresses, subjects, bodies, prompts, or answers.
|
||||
|
||||
Post-fix representative direct/follow-up coverage also passes for every family:
|
||||
notes/calendar 4/4, tasks/documents/memory/skills/Cookbook/search/shell 14/14,
|
||||
and email 3/3 in its privacy-preserving runner. The combined legacy verifier's
|
||||
metadata-only email guard correctly refused its broader stable inventory; that
|
||||
stopped report is not counted as a model failure.
|
||||
|
||||
Evidence:
|
||||
|
||||
- `reports/clean-v3-stateful-all-r3-20260909.json`
|
||||
- `reports/clean-v3-stateful-documents-r2-20260909.json`
|
||||
- `reports/clean-v3-search-toggle-final-r6-20260909.json`
|
||||
- `reports/clean-v3-search-quality-r3-20260909.json`
|
||||
- `reports/clean-v3-email-read-r1-20260909.json`
|
||||
- `reports/clean-v3-ten-family-tail-postfix-r1-20260909.json`
|
||||
|
||||
These checks verify routing, execution, persistence, follow-up, and selected
|
||||
answer-quality invariants. They are not yet the sealed all-action ship score.
|
||||
|
||||
## Compact v5 and corrected contract evidence
|
||||
|
||||
Compact v5 keeps the compact-v3 surface and adds only development-positive
|
||||
field hints for Email, Search/Hugging Face quant selection, and Shell/files.
|
||||
A Calendar date hint regressed development and was excluded. The Python tool now
|
||||
emits one final bare expression, REPL-style, without duplicating explicit
|
||||
`print(...)`; this turns otherwise correct computation calls into visible tool
|
||||
evidence for all models.
|
||||
|
||||
Under frozen scorer `odysseus.contract.v2.5`, development is 327/344 raw
|
||||
(95.06%) and 327/336 scorable (97.32%). Sealed blind is 311/344 raw (90.41%)
|
||||
and 311/336 scorable (92.56%), with zero reasoning leakage. Calendar, Shell,
|
||||
and Tasks remain below the 90% family ship floor, so the model is not yet a
|
||||
full benchmark ship candidate.
|
||||
|
||||
Fresh post-deploy real-UI evidence passes: stateful flows 6/6, Email 3/3,
|
||||
Search quality/recovery 3/3, private browser 3/3, and VL workflow 3/3. The
|
||||
Search check accepts a failed attempt only when a later tool succeeds and the
|
||||
final answer remains grounded.
|
||||
@@ -0,0 +1,125 @@
|
||||
# Regular-model tool compatibility
|
||||
|
||||
Last verified: 2026-09-09 through the authenticated 7011 Agent UI as
|
||||
`sft_alex_creator`.
|
||||
|
||||
This is the legacy-RAG track. The exact model
|
||||
`odysseus-qwen3.5-tools-pre-heretic` is excluded and remains on its model-owned
|
||||
clean compact runtime.
|
||||
|
||||
## Current baseline
|
||||
|
||||
| Endpoint | Model | Ten-family result | State |
|
||||
|---|---|---:|---|
|
||||
| DeepSeek | `deepseek-v4-flash` | 10/10 | passed |
|
||||
| DeepSeek | `deepseek-v4-pro` | 10/10 | passed |
|
||||
| OpenAI | `gpt-5.5` | 10/10 | passed |
|
||||
| OpenAI | `gpt-5.6-sol` | 10/10 | passed |
|
||||
| OpenAI | `gpt-5.6-terra` | 10/10 | passed |
|
||||
| OpenAI | `gpt-5.6-luna` | 10/10 | passed |
|
||||
| OpenRouter | `moonshotai/kimi-k3` | 10/10 | passed |
|
||||
| OpenRouter | `x-ai/grok-4.5` | 10/10 | passed |
|
||||
| OpenRouter | `qwen/qwen3-vl-235b-a22b-instruct` | 10/10 | passed |
|
||||
| OpenRouter | `openai/gpt-5-image` | n/a | image generation; chat tools unsupported |
|
||||
| Local `100.69.120.65:8062` | `Qwen/Qwen3.5-9B` | not run | endpoint unavailable |
|
||||
| Local `100.69.120.65:8062` | `GLM-5.3-Flash-Alis-MLX-4bit` | not run | endpoint unavailable |
|
||||
|
||||
The ten-family baseline covers one read-only functional turn each for notes,
|
||||
calendar, email accounts, tasks, documents, memory, skills, Cookbook/admin,
|
||||
web search, and shell. It verifies the legacy route, expected native tool call,
|
||||
execution result, visible UI answer, and absence of reasoning leakage. It is not
|
||||
yet a claim that every mutation/action variant, typo, or follow-up passes.
|
||||
|
||||
## Typo and follow-up profile
|
||||
|
||||
The stricter real-UI profile sends one misspelled read-only request to every
|
||||
family, followed immediately by a noun-free reference to the returned result.
|
||||
Read-only follow-ups must not call any tool; search follow-ups may either use
|
||||
the existing evidence or fetch the prior link. Across the nine chat-capable API
|
||||
models, the composited post-repair result is **178/180 turns (98.89%)**:
|
||||
|
||||
| Model | Conversation result |
|
||||
|---|---:|
|
||||
| `deepseek-v4-flash` | 20/20 |
|
||||
| `deepseek-v4-pro` | 20/20 |
|
||||
| `gpt-5.5` | 20/20 |
|
||||
| `gpt-5.6-sol` | 20/20 |
|
||||
| `gpt-5.6-terra` | 20/20 |
|
||||
| `gpt-5.6-luna` | 18/20 |
|
||||
| `moonshotai/kimi-k3` | 20/20 |
|
||||
| `x-ai/grok-4.5` | 20/20 |
|
||||
| `qwen/qwen3-vl-235b-a22b-instruct` | 20/20 |
|
||||
|
||||
Luna's only remaining family miss is a deliberately misspelled Shell request.
|
||||
The correct-spelling baseline passes. The harness does not auto-execute a shell
|
||||
command to hide that model-owned limitation.
|
||||
|
||||
The shared repair recognizes a uniquely misspelled action verb and family noun,
|
||||
then seals only declared safe private reads with immutable canonical arguments.
|
||||
This repaired Tasks/Documents/Memory and adjacent read families across providers
|
||||
without widening mutation or Shell authority. A compact native-tool instruction
|
||||
also tells regular API models to map clear typos to a currently offered tool.
|
||||
|
||||
Conversation evidence:
|
||||
|
||||
- `reports/regular-model-conversation-flash-r3-20260909.json`
|
||||
- `reports/regular-model-conversation-remaining-r1-20260909.json`
|
||||
- `reports/regular-model-conversation-repair-r1-20260909.json`
|
||||
- `reports/regular-model-conversation-shell-r1-20260909.json`
|
||||
- `reports/regular-model-conversation-qwen-repair-r1-20260909.json`
|
||||
- `reports/regular-model-conversation-qwen-tail-r1-20260909.json`
|
||||
- `reports/regular-model-conversation-qwen-search-r1-20260909.json`
|
||||
|
||||
Evidence:
|
||||
|
||||
- `reports/regular-model-tools-provider-final-r4-20260909.json` — Flash, GPT-5.5, Kimi: 30/30.
|
||||
- `reports/regular-model-tools-repair-r3-20260909.json` — Pro and Sol: 20/20; retained Qwen pre-final 9/10 miss.
|
||||
- `reports/regular-qwen-vl-full-r4-20260909.json` — Qwen-VL final family-switch run: 10/10.
|
||||
- `reports/regular-model-tools-remaining-20260909.json` — Terra, Luna, Grok: 30/30; records unavailable/unsupported models and pre-repair failures.
|
||||
- `reports/regular-model-tools-postfix-r1-20260909.json` — post-hardening
|
||||
rerun: nine chat-capable API models passed 90/90 family turns with zero model
|
||||
failures. Its overall status is non-passing only because the two configured
|
||||
local endpoints were offline; the image-only model remains unsupported.
|
||||
|
||||
## Family switch and page inspection
|
||||
|
||||
The six-turn switch/back flow covers notes → calendar → notes from prior
|
||||
evidence → web search → explicit `web_fetch` → calendar from prior evidence.
|
||||
All nine API models have a clean 6/6 reproduction (**54/54**). Kimi skipped
|
||||
search once in the retained first run and passed a fresh reproduction; that
|
||||
variability remains visible instead of being erased.
|
||||
|
||||
Evidence:
|
||||
|
||||
- `reports/regular-model-switchback-flash-r2-20260909.json`
|
||||
- `reports/regular-model-switchback-remaining-r1-20260909.json`
|
||||
- `reports/regular-model-switchback-kimi-r1-20260909.json`
|
||||
|
||||
## Repair that produced the clean baseline
|
||||
|
||||
Regular models no longer inherit up to three stale tool families into every
|
||||
explicit new request. Referential follow-ups still resolve from typed recent
|
||||
tool evidence, while explicit family switches receive the current family only.
|
||||
Safe required reads use `active_capabilities`, so stale offered context cannot
|
||||
disable their immutable operation. The stream layer also stops an exact long
|
||||
block repeated twice instead of waiting for a provider's full timeout.
|
||||
|
||||
The composer no longer treats generic words such as “source”, “system”, “app”,
|
||||
or “review” as authority to silently enable Bash. Explicit shell, terminal,
|
||||
repository, code-file, and direct coding requests retain workspace
|
||||
auto-escalation. This is a shared UI authority fix, not a model-name exception.
|
||||
|
||||
Run a bounded subset with:
|
||||
|
||||
```sh
|
||||
MODELS='deepseek-v4-flash,gpt-5.5' \
|
||||
FAMILIES='notes,calendar' WORKERS=2 \
|
||||
REPORT_PATH=reports/regular-model-check.json \
|
||||
node scripts/verify_regular_model_tools.mjs
|
||||
```
|
||||
|
||||
Set `PROFILE=conversation` to run the typo plus follow-up profile.
|
||||
|
||||
The runner discovers only enabled pinned models (visible cached local models
|
||||
when no pins exist), retains no tool outputs or private rows, and deletes only
|
||||
the exact sessions it creates.
|
||||
@@ -0,0 +1,92 @@
|
||||
# Search and compact-tool experiment — 2026-09-09
|
||||
|
||||
## Decision
|
||||
|
||||
Keep the normal routed profile on 7011. The all-tools compact experiment is
|
||||
implemented but **disabled**: direct routing success did not translate into a
|
||||
working Agent UI. Do not retrain or promote a profile on these measurements.
|
||||
|
||||
## Changes
|
||||
|
||||
- Short public-web lookups on the target model have an execution budget: two
|
||||
distinct token-normalized searches, one fetch, and up to three browser calls
|
||||
after the two searches. This bounds attempts, not just recovery prose. Existing
|
||||
permissions still apply; this does not make unavailable tools executable.
|
||||
- Failed/weak searches reach the model for evaluation and query refinement,
|
||||
instead of the earlier unconditional terminal evidence veto. Some legacy
|
||||
heuristics and official-site shortcuts remain; this is not a completed rewrite.
|
||||
- Search providers retain query/engine/date provenance. Unconfigured credentialed
|
||||
fallbacks are skipped. When SearXNG is the sole configured usable provider, Yep
|
||||
on the same instance is an additional fallback.
|
||||
- An unavailable warm-only family no longer vetoes an otherwise ordinary reply.
|
||||
- The all-tools experiment offers the trained compact inventory subject to
|
||||
permissions. It requires the exact test-owner environment flag and exact model
|
||||
match. The temporary service flag was removed after failed UI testing.
|
||||
|
||||
## Evidence and limits
|
||||
|
||||
| Measurement | Result | What it establishes |
|
||||
|---|---|---|
|
||||
| Focused Python regression suite | 401 passed | Covered policy, contract, provider and recovery-budget behavior |
|
||||
| Direct family-only compact schemas | 10/10 tool routing | Small public smoke test, not functional or blind accuracy |
|
||||
| Direct all-family compact schemas | 10/10 tool routing | Inventory did not break these first calls; roughly 3–4x slower in this run |
|
||||
| Full compact Agent UI, revision 3 | All eight turns failed one or more checks | Not suitable for activation; leaks, duplicate/incorrect rendering or missing expected calls |
|
||||
| Normal-profile final UI control | Five passed checks, two product failures, one capture error | Not accepted; suite status incomplete |
|
||||
| Clean synthetic notes tool-result continuation | Clean answer with both schema sizes | Model can continue correctly on that isolated input, not proof that UI failure is solely harness |
|
||||
|
||||
Direct probes used temperature 0; the UI target-model sampling path can cap at
|
||||
0.2. Prompts, history and tool-result serialization also differ. Match those
|
||||
before attributing UI failures to weights versus harness. Existing UI checks are
|
||||
not a grounded factual-answer benchmark. Unit tests are not UI acceptance.
|
||||
|
||||
Normal-profile control details: notes initial, both calendar turns, AI search
|
||||
initial, and history initial passed the automated checks. Notes follow-up hit a
|
||||
Playwright `Network.getResponseBody` capture error and is inconclusive. AI search
|
||||
follow-up explicitly requested a summary with no tools, but the contract still
|
||||
required `search_browser` and returned a permission failure. The history search
|
||||
follow-up failed the visible-leak check. These are separate from source relevance;
|
||||
the earlier warm-only fix did not cover classification as an active requirement.
|
||||
There is no matched pre-change control establishing a net improvement.
|
||||
|
||||
Provider isolation bypassed app relevance filters. Bing general often returned
|
||||
broad or unrelated results despite the full query. Google/Mojeek returned no
|
||||
results, DDG hit CAPTCHA, and Presearch timed out. Yep returned useful PostgreSQL
|
||||
documentation, but was weak or empty for several other questions. Engine health
|
||||
and source quality remain unresolved. Fallback cannot help when an earlier weak
|
||||
result survives filtering; there is no claim of universal relevance here.
|
||||
|
||||
## Reproduce and inspect
|
||||
|
||||
- `scripts/audit_search_pipeline.py`: raw provider comparison, no model.
|
||||
- `scripts/compare_compact_tool_inventory.py`: read-only model schema comparison;
|
||||
proposed calls are never executed.
|
||||
- `scripts/verify_agent_turn_contract.mjs`: real 7011 Agent UI and persisted-history
|
||||
checks. Use the dedicated test account; reports can contain private tool data.
|
||||
- `reports/search-provider-isolation.json`, `reports/search-yep-isolation.json`:
|
||||
public provider evidence.
|
||||
- `reports/compact-inventory-ablation.json`: direct routing probe.
|
||||
- `reports/full-compact-ui-audit-r3.json`: completed rejected UI experiment.
|
||||
Earlier experiment reports include an initialization error and an aborted run;
|
||||
do not combine them into an accuracy score.
|
||||
- `reports/routed-control-ui-audit-final.json`: normal-profile control replay;
|
||||
eight attempts, incomplete because of the capture error; failures retained.
|
||||
|
||||
Focused suite:
|
||||
|
||||
```sh
|
||||
/home/pewds/odysseus-cookbook-fresh/.venv/bin/pytest -q tests/test_turn_contract.py tests/test_turn_contract_integration.py tests/test_service_search_provider_guards.py tests/test_web_recovery_budget.py tests/test_tool_policy.py
|
||||
```
|
||||
|
||||
UI replay (read-only prompts, creates test chats):
|
||||
|
||||
```sh
|
||||
node scripts/verify_agent_turn_contract.mjs --families notes,calendar,search_ai,search_history --pairs notes:11,calendar:11,search_ai:11,search_history:11 --max-turns 8 --total-ms 360000 --turn-ms 45000 --report reports/routed-control-ui-audit-final.json
|
||||
```
|
||||
|
||||
## Next discriminating test
|
||||
|
||||
Replay the same captured UI request directly, preserving sampling, compact
|
||||
schemas, history and tool results. Then change one layer at a time. Separately
|
||||
score retrieved-source relevance and supported answers. Replace failing generic
|
||||
boundaries only when the replay identifies them; do not add rules for individual
|
||||
user phrasings or treat successful tool routing as successful execution.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Typo-tolerant tool routing audit
|
||||
|
||||
The 9B SFT model was not retrained. This audit targets the earlier harness
|
||||
stage that decides which complete tool families the model is allowed to see.
|
||||
|
||||
## Method
|
||||
|
||||
- Source prompts: real `sft_alex_creator` sessions from `a37dcb3b-...` onward.
|
||||
- Labels: recorded single-family tool calls, excluding mixed/ambiguous traces.
|
||||
- Variants: deletion, adjacent transposition, duplicated character,
|
||||
keyboard-neighbor substitution, and accidental word split.
|
||||
- Split: deterministic SHA-256 assignment before scoring (75% dev, 25% blind).
|
||||
- Safety: static routing only; no historical mutation or send action is replayed.
|
||||
- Acceptance: at least 95% blind exact-family accuracy and below 1% blind
|
||||
wrong-family authorization. Abstention is measured separately.
|
||||
|
||||
## Results
|
||||
|
||||
| Router | Dev family supplied | Blind family supplied | Blind exact | Blind wrong-family |
|
||||
|---|---:|---:|---:|---:|
|
||||
| Previous exact rules | 63.64% | 65.69% | — | — |
|
||||
| Conservative fuzzy fallback r4 | 96.31% | 98.31% | 96.62% | 0.00% |
|
||||
| Final router + safe-read repair | 98.31% | 98.73% | 97.05% | 0.00% |
|
||||
|
||||
The fallback runs only for action/lookup-shaped requests, resolves exactly one
|
||||
nearby family term, and abstains on ambiguity. Conceptual questions remain
|
||||
tool-free. Complete family schemas are still selected by the immutable turn
|
||||
contract; fuzzy matching never chooses an individual tool or its arguments.
|
||||
|
||||
Authoritative machine reports:
|
||||
|
||||
- `reports/typo-tool-routing-baseline-20260909.json`
|
||||
- `reports/typo-tool-routing-fuzzy-r4-20260909.json`
|
||||
- `reports/typo-tool-routing-final-20260909.json`
|
||||
- `reports/post-followup-agent-80-20260909.json`
|
||||
- `reports/post-typo-routing-agent-80-20260909.json`
|
||||
- `reports/live-typo-agent-20-20260909.json`
|
||||
- `reports/live-typo-unresolved-r3-20260909.json`
|
||||
- `reports/live-typo-agent-final-20-20260909.json`
|
||||
- `reports/post-typo-safe-read-agent-final-80-20260909.json`
|
||||
|
||||
## Live 7011 findings
|
||||
|
||||
The post-deployment standard matrix passed 80/80 through the real Agent UI.
|
||||
The first read-only typo matrix then attempted 17 of 20 planned turns before
|
||||
its total-time limit. Initial Notes, Calendar, Email, Tasks, Documents, and
|
||||
Cookbook calls passed. Completed failing turns still had the correct family
|
||||
and required tool in `turn_contract.offered`; the 9B model sometimes answered
|
||||
without calling that offered tool. Memory and Search also exposed timeouts.
|
||||
|
||||
This separates three failure classes:
|
||||
|
||||
1. **Tool injection:** addressed by conservative fuzzy family routing; blind
|
||||
exact routing is 96.62% with zero blind wrong-family authorizations.
|
||||
2. **Required read execution:** a correctly offered safe list/refresh tool can
|
||||
still be skipped by the model, especially after a typo or on “list those
|
||||
again” follow-ups. This should be handled by the generic deterministic
|
||||
safe-read path, not additional prompt-specific hints.
|
||||
3. **Runtime timeout:** Search and one Memory follow-up require loop/backend
|
||||
diagnosis. A timeout is not counted as a model-accuracy or routing result.
|
||||
|
||||
The generic safe-read parser and search-family precedence were then repaired.
|
||||
The previously unresolved Calendar, Email, Search, and Shell/Files cases passed
|
||||
8/8. The complete typo matrix passed 20/20, including initial requests and
|
||||
follow-ups for all ten families. The final standard Agent UI compatibility
|
||||
matrix passed 80/80 across family, Web-toggle, and follow-up combinations.
|
||||
|
||||
The broad routing regression suite passed 458 tests. The model was not
|
||||
retrained and no DeepSeek API was used: the measured defect was in harness
|
||||
family selection and deterministic safe-read execution, upstream of the
|
||||
model. All 1,535 unique labeled historical turns were statically audited to
|
||||
mine failure categories. Historical write/send/delete actions were not replayed
|
||||
against live data; live verification used the deduplicated read-only matrices.
|
||||
@@ -0,0 +1,26 @@
|
||||
# Skills lifecycle
|
||||
|
||||
The UI exposes All, Built-in, Approved, and Draft. Draft includes archived
|
||||
records so they remain inspectable and recoverable. Built-ins are not audited.
|
||||
Approved means published, passing, at the configured confidence threshold,
|
||||
and not marked unnecessary. Baseline speed measurements remain evidence, not
|
||||
an additional hidden UI approval gate.
|
||||
|
||||
Automatic audits process at most eight eligible records at a time, oldest first.
|
||||
New records are eligible immediately; inconclusive checks retry after a day;
|
||||
failed repairs retry after a week. Passed, duplicate-skipped, and archived records
|
||||
are excluded. Existing daily Skills Audit tasks drive this queue. Their quiet
|
||||
window deferrals propagate to the scheduler rather than becoming task failures.
|
||||
Automatic runs use background model scheduling. Existing self-repair and teacher
|
||||
repair stages remain in place; failed candidates remain drafts.
|
||||
|
||||
The skill index advertises short descriptions; the agent loads a relevant full
|
||||
procedure on demand and applies already-injected procedures directly. Extraction
|
||||
prefers verified discoveries and specific workarounds over routine tool usage.
|
||||
|
||||
Reference reviewed: NousResearch/hermes-agent, MIT license, commit
|
||||
cfdbbb6e35010ace89fbe8243ee82fa4de143e10, cloned to
|
||||
/home/pewds/hermes-skills-reference. In particular tools/skills_tool.py and
|
||||
agent/prompt_builder.py use progressive disclosure and task-triggered procedure
|
||||
loading. These changes adapt that approach to Odysseus's existing registry;
|
||||
no Hermes implementation code was copied.
|
||||
@@ -163,6 +163,10 @@ if (Test-Path $cudaBase) {
|
||||
}
|
||||
|
||||
# 7. Start the server (use `python -m uvicorn` - bare `uvicorn` may not be on PATH)
|
||||
# -Port only reaches uvicorn as a flag. Everything that builds a URL for this
|
||||
# instance - internal_api_base(), companion pairing, the MCP OAuth callback -
|
||||
# reads APP_PORT, so set it too or they all assume 7000.
|
||||
$env:APP_PORT = $Port
|
||||
Write-Step ("Starting Odysseus at http://{0}:{1}" -f $BindHost, $Port)
|
||||
Write-Host "Press Ctrl+C to stop."
|
||||
Write-Host ""
|
||||
|
||||
+1
-1
@@ -130,7 +130,7 @@ if __name__ == "__main__":
|
||||
from app import app
|
||||
|
||||
bind_host = os.getenv("APP_BIND", "127.0.0.1")
|
||||
bind_port = int(os.getenv("APP_PORT", "7000"))
|
||||
bind_port = int(os.getenv("APP_PORT", "7011"))
|
||||
url = f"http://{bind_host}:{bind_port}"
|
||||
|
||||
if getattr(sys, 'frozen', False):
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
The MIT License (MIT)
|
||||
|
||||
Copyright (c) 2013-2020 Khan Academy and other contributors
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,21 @@
|
||||
The MIT License (MIT)
|
||||
|
||||
Copyright (c) 2014 - 2022 Knut Sveidqvist
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
+2370
-75
File diff suppressed because it is too large
Load Diff
@@ -81,7 +81,13 @@ async def call_tool(name: str, arguments: dict) -> list[TextContent]:
|
||||
if not model_spec:
|
||||
return [TextContent(type="text", text="Error: No image model found. Configure one in Admin.")]
|
||||
|
||||
url, model_id, headers = await asyncio.to_thread(_resolve_model, model_spec)
|
||||
try:
|
||||
url, model_id, headers = await asyncio.to_thread(_resolve_model, model_spec, model_type="image")
|
||||
except ValueError:
|
||||
_lower_model_spec = model_spec.lower()
|
||||
if not any(_name in _lower_model_spec for _name in ("gpt-image", "dall-e")):
|
||||
raise
|
||||
url, model_id, headers = await asyncio.to_thread(_resolve_model, model_spec)
|
||||
|
||||
is_gpt_image = "gpt-image" in model_id.lower()
|
||||
base_url = url.replace("/chat/completions", "").replace("/v1/messages", "").rstrip("/")
|
||||
|
||||
@@ -17,6 +17,8 @@ from mcp.types import Tool, TextContent
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from src.memory import MemoryStoreUnreadable
|
||||
|
||||
server = Server("memory")
|
||||
|
||||
# Late-initialized managers (set during first tool call)
|
||||
@@ -29,6 +31,10 @@ _OWNER_SCOPE_ERROR = (
|
||||
"Error: Memory MCP owner is not configured for an owner-scoped memory store. "
|
||||
"Set ODYSSEUS_MCP_MEMORY_OWNER for this server or use the owner-aware native memory tool."
|
||||
)
|
||||
_UNREADABLE_STORE_ERROR = (
|
||||
"Error: Memory store is temporarily unreadable — nothing was saved. "
|
||||
"Repair or restore memory.json, then retry."
|
||||
)
|
||||
|
||||
|
||||
def _configured_owner() -> str | None:
|
||||
@@ -51,9 +57,21 @@ def _owner_scoped_store(entries: list[dict]) -> bool:
|
||||
return any(_entry_owner(entry) for entry in entries if isinstance(entry, dict))
|
||||
|
||||
|
||||
def _scope_entries() -> tuple[str | None, list[dict], list[dict], str | None]:
|
||||
"""Return configured owner, all entries, visible entries, and optional error."""
|
||||
entries = _memory_manager.load_all()
|
||||
def _scope_entries(for_update: bool = False) -> tuple[str | None, list[dict], list[dict], str | None]:
|
||||
"""Return configured owner, all entries, visible entries, and optional error.
|
||||
|
||||
``for_update=True`` is for read-modify-write callers. They save the ``all
|
||||
entries`` list back, so an unreadable store must be reported as an error
|
||||
instead of degrading to ``[]`` — otherwise the save writes their one new
|
||||
entry over the whole store (issue #5673).
|
||||
"""
|
||||
if for_update:
|
||||
try:
|
||||
entries = _memory_manager.load_all_for_update()
|
||||
except MemoryStoreUnreadable as e:
|
||||
return None, [], [], f"{_UNREADABLE_STORE_ERROR} ({e})"
|
||||
else:
|
||||
entries = _memory_manager.load_all()
|
||||
owner = _configured_owner()
|
||||
if owner is None and _owner_scoped_store(entries):
|
||||
return None, entries, [], _OWNER_SCOPE_ERROR
|
||||
@@ -161,7 +179,7 @@ async def call_tool(name: str, arguments: dict) -> list[TextContent]:
|
||||
category = arguments.get("category", "fact")
|
||||
if not text:
|
||||
return _text_result("Error: Memory text cannot be empty")
|
||||
owner, memories, _visible, scope_error = _scope_entries()
|
||||
owner, memories, _visible, scope_error = _scope_entries(for_update=True)
|
||||
if scope_error:
|
||||
return _text_result(scope_error)
|
||||
entry = _memory_manager.add_entry(text, source="ai_agent", category=category, owner=owner)
|
||||
|
||||
Generated
+69
-4
@@ -4,19 +4,84 @@
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "odysseus",
|
||||
"devDependencies": {
|
||||
"@antithesishq/bombadil": "^0.6.1"
|
||||
"@antithesishq/bombadil": "^0.7.0",
|
||||
"@playwright/test": "^1.62.1"
|
||||
}
|
||||
},
|
||||
"node_modules/@antithesishq/bombadil": {
|
||||
"version": "0.6.1",
|
||||
"resolved": "https://registry.npmjs.org/@antithesishq/bombadil/-/bombadil-0.6.1.tgz",
|
||||
"integrity": "sha512-d1iufG3MI7gSMSiSmMeNdcMW+qR0yQXL2zdkVynC3n3DYgFJYlYXKUQzygmqU12m4RWlR5iOdQU1hsx5UT6+IA==",
|
||||
"version": "0.7.0",
|
||||
"resolved": "https://registry.npmjs.org/@antithesishq/bombadil/-/bombadil-0.7.0.tgz",
|
||||
"integrity": "sha512-alJmnphJ/iUoL5mCsnV3DwtajGy/sEQ3NJJCiMhgjqXshSq2BUtAs0vqdXEiiSkB8HbsOX5CLrAcaogYdwfAJg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"bin": {
|
||||
"bombadil": "bin/bombadil.js"
|
||||
}
|
||||
},
|
||||
"node_modules/@playwright/test": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.62.1.tgz",
|
||||
"integrity": "sha512-DTcUc8qii+cpHvtOwggMtBRMjKZHXYWdw8syRYu2vtzuq4Wxphqq4NfCs5Zt44L6mA8rfDfj+PHnxFc/FeK6mQ==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"playwright": "1.62.1"
|
||||
},
|
||||
"bin": {
|
||||
"playwright": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
}
|
||||
},
|
||||
"node_modules/fsevents": {
|
||||
"version": "2.3.2",
|
||||
"resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz",
|
||||
"integrity": "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==",
|
||||
"dev": true,
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": "^8.16.0 || ^10.6.0 || >=11.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/playwright": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/playwright/-/playwright-1.62.1.tgz",
|
||||
"integrity": "sha512-0M+L3LAD8/nm554LOla9Ayx0j0tmFZ0FBcoQ7F1VuVHpM/XpiC8RcDzBQB8W5+hA8L22THxELzeF+2WcUzvcLg==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"playwright-core": "1.62.1"
|
||||
},
|
||||
"bin": {
|
||||
"playwright": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"fsevents": "2.3.2"
|
||||
}
|
||||
},
|
||||
"node_modules/playwright-core": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.62.1.tgz",
|
||||
"integrity": "sha512-wPYSwEBJY9GHraISXqyqtx0na0LpO3XEX7jNDhntbex7tzUS7kLnZsOlFruFJB4Hi/rhDMjXGqHewDZ68nYZVw==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"bin": {
|
||||
"playwright-core": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+11
-2
@@ -1,9 +1,18 @@
|
||||
{
|
||||
"name": "odysseus",
|
||||
"private": true,
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "https://github.com/pewdiepie-archdaemon/odysseus.git"
|
||||
"url": "https://github.com/odysseus-dev/odysseus.git"
|
||||
},
|
||||
"scripts": {
|
||||
"test:photo-editor": "playwright test --config tests/e2e/playwright.config.js",
|
||||
"test:photo-editor:install": "playwright install chromium firefox webkit",
|
||||
"test:photo-editor:firefox": "PHOTO_EDITOR_E2E_BROWSER=firefox playwright test --config tests/e2e/playwright.config.js",
|
||||
"test:photo-editor:webkit": "PHOTO_EDITOR_E2E_BROWSER=webkit playwright test --config tests/e2e/playwright.config.js"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@antithesishq/bombadil": "^0.6.1"
|
||||
"@antithesishq/bombadil": "^0.7.0",
|
||||
"@playwright/test": "^1.62.1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,326 @@
|
||||
# Odysseus Tool Runtime Hardening Plan
|
||||
|
||||
## Objective
|
||||
|
||||
Ship `odysseus-qwen3.5-tools-pre-heretic` with one compact, model-specific tool
|
||||
runtime that supports realistic multi-turn use. Keep the existing RAG runtime
|
||||
unchanged for every other model. Prove routing, execution, answer quality,
|
||||
follow-ups, safety, rendering, latency, and native image/VL understanding through
|
||||
the real 7011 Agent UI.
|
||||
|
||||
Current evidence is a baseline, not a ship claim:
|
||||
|
||||
- Corrected v2.5 + compact-v5 development is 327/344 raw (95.06%) and
|
||||
327/336 scorable (97.32%). Sealed blind is 311/344 raw (90.41%) and
|
||||
311/336 scorable (92.56%), with zero reasoning leakage.
|
||||
- Notes, Skills, and Cookbook/admin clear 95% scorable blind. Calendar 87.5%,
|
||||
Shell/files 86.11%, and Tasks 87.5% remain below the 90% family ship floor.
|
||||
- Compact-v5 hints improved Email, Search/HF quant, and Shell on development;
|
||||
a Calendar hint regressed and was rejected rather than shipped.
|
||||
|
||||
- Ten-family focused baseline: 19/20 functional and 20/20 routing/execution.
|
||||
- Typo and cross-family read flows: 26/26 passed.
|
||||
- Real use exposed untested write correction and search-to-fetch follow-ups.
|
||||
- Email production access, browser interaction, search quality, and broader
|
||||
multi-turn mutations are not yet proven.
|
||||
- Nine enabled chat-capable regular API models pass the ten-family read-only
|
||||
legacy-RAG baseline (90/90 combined). Their stricter typo/follow-up profile is
|
||||
178/180 turns: eight models are 20/20 and Luna is 18/20 due only to its
|
||||
misspelled Shell request. One pinned image-generation model is explicitly
|
||||
unsupported and two visible local models are currently offline.
|
||||
- Native VL object/spatial recognition and reload follow-up pass. Exact OCR
|
||||
fails equally on the fine-tune and untouched 9B base and remains unresolved.
|
||||
PNG, JPEG, and WebP transport all pass.
|
||||
- Reversible create/correct/API-verify/cleanup flows pass 6/6 across every
|
||||
stateful family.
|
||||
- Search Web-toggle combinations pass 8/8 and the focused quality suite passes
|
||||
3/3. Production-path email account/inbox/referential reads pass 3/3.
|
||||
- The latest regular-model regression is 90/90 across the nine enabled
|
||||
chat-capable API models, with zero failed model turns; two local endpoints
|
||||
remain offline and the image-only model is unsupported.
|
||||
- The Epictetus OMLX endpoint was recovered after an unsupported
|
||||
`qwen3_5_mtp` model load wedged the server. Its supported Qwen 27B 4-bit
|
||||
model passes the ten-family real-7011 legacy-RAG smoke 10/10; the unsupported
|
||||
MTP artifact is recorded as a runtime limitation rather than a timeout.
|
||||
- Fresh compact-v5 UI regressions pass stateful 6/6, Email 3/3, Search 3/3,
|
||||
private-browser 3/3, and VL workflow 3/3.
|
||||
- The exact-model, family-scoped compact runtime now passes 20/20 direct and
|
||||
same-family turns across all ten families on the real 7011 Agent UI. A
|
||||
separate 36/36 robustness run passes misspellings, bounded repeats, browser
|
||||
and news continuation, ambiguous follow-ups, family switchbacks, and a
|
||||
greeting before a tool request.
|
||||
- The mobile active-email editor path passes 1/1: `Write reply this email`
|
||||
offers and executes only `update_document`, mutates the open draft, and
|
||||
preserves its reply headers and quoted thread.
|
||||
- The active-editor classifier now also covers short mobile wording without a
|
||||
pronoun (`Write reply` / `Draft a reply`) while explicit note, code, file, and
|
||||
new-object requests retain their own families. Whole-draft requests are bound
|
||||
to the sole offered `update_document` writer until one successful write, then
|
||||
tools are removed for the confirmation round. The deployed real-route email
|
||||
regression passes 3/3—including the exact unspecified `Write reply to this
|
||||
email` form—with one write, verified mutation, and preserved reply headers.
|
||||
Clean-v3 now also emits the established `doc_update` event and flattened
|
||||
document metadata on `tool_output`, so a successful database write updates
|
||||
the already-open editor instead of leaving stale UI beside a success message.
|
||||
- The client now reuses the existing assistant bubble for `agent_step` round 1
|
||||
instead of replacing it before the first token. A real-7011 sampled
|
||||
greeting-to-Notes conversation passes 2/2 with stable first-round DOM
|
||||
identity; round 2+ remains the only continuation-bubble path.
|
||||
- Clean-runtime metrics now expose provider-counted initial injected tokens,
|
||||
all-round input/output, TTFT, tok/s, schema count, agent rounds, and tool-call
|
||||
count. A real 7011 browser run passes 2/2 and visibly renders compact footers
|
||||
plus the full details popup; the sampled Notes turns streamed progressively.
|
||||
- The deployed startup bottleneck was an unindexed quadratic transcript-FTS
|
||||
reconciliation. Live-database import fell from about 36 seconds to 0.54
|
||||
seconds; 7011 now answers in about 3 seconds after a controlled restart.
|
||||
- A controlled identical-compact comparison already proves the fine-tune's
|
||||
accuracy benefit: 94.48% (325/344) versus the untouched base's 77.91%
|
||||
(268/344). Raw serving speed is effectively tied, so product speed comes
|
||||
from the compact contract and fewer failed/redundant rounds.
|
||||
- A fully merged 10,000-row category-repair candidate reached 97.32% scorable
|
||||
development but only 92.26% scorable sealed blind. Calendar (87.5%), Tasks
|
||||
(87.5%), and Shell/files (86.11%) remained below the family floor, so it was
|
||||
rejected and not deployed. Compact-v4/full development A/Bs did not improve
|
||||
Calendar or Tasks over compact-v5; full-schema Shell also fell from 97.22%
|
||||
to 94.44%. This rules out compactness as the primary cause of the remaining
|
||||
blind gaps and supports keeping the compact contract.
|
||||
|
||||
## Non-negotiable architecture rules
|
||||
|
||||
1. Runtime selection follows exact model identity. The trained Odysseus model
|
||||
uses the clean compact runtime across endpoint aliases; all other models use
|
||||
legacy RAG. Add a regression test for both sides.
|
||||
2. Resolve permissions, toggles, and available backends once per turn. Produce
|
||||
one immutable contract satisfying `required ⊆ offered ⊆ executable`.
|
||||
3. Never offer a tool that the preview policy will categorically reject. Add a
|
||||
contract self-check covering every offered action/effect combination.
|
||||
4. Follow-ups consume typed prior evidence: native call, result, success state,
|
||||
family, and object identifiers. Do not infer continuity from keyword RAG.
|
||||
5. Contextual write authority may revise only a recently proven object in the
|
||||
same family. It may not authorize a new object, another family, a destructive
|
||||
action, or an external side effect.
|
||||
6. The model chooses tools and valid arguments. The harness validates and
|
||||
executes; it does not silently substitute another family, rewrite arguments,
|
||||
fabricate success, or replace a failed tool with prose claiming completion.
|
||||
7. One owner renders each turn: streamed prose or canonical structured output.
|
||||
Never both, and never expose hidden prompts or raw untrusted wrappers.
|
||||
8. No exact-prompt production patches. A fix must name the failed layer, add a
|
||||
generic failing invariant test, and cover neighboring cases.
|
||||
|
||||
## Failure layers
|
||||
|
||||
Every failure is assigned to exactly one primary layer before code changes:
|
||||
|
||||
1. **Route:** wrong model runtime or endpoint identity.
|
||||
2. **Contract:** required tool absent, forbidden tool present, or toggle drift.
|
||||
3. **Model:** wrong/no tool or semantically wrong required arguments despite a
|
||||
correct contract.
|
||||
4. **Policy:** valid proposed operation incorrectly allowed or denied.
|
||||
5. **Execution:** canonical arguments, backend dispatch, timeout, or result
|
||||
envelope is wrong.
|
||||
6. **Evidence:** result is empty, irrelevant, truncated badly, or insufficient.
|
||||
7. **Answer:** model misstates or ignores valid tool evidence.
|
||||
8. **Rendering:** duplicate, dump-at-end, missing structured output, or stopped
|
||||
stream.
|
||||
9. **Performance:** startup, TTFT, tool latency, or oversized context.
|
||||
|
||||
Reports store aggregate category, relevant contract/tool metadata, timings, and
|
||||
sanitized outputs. Do not copy private hidden benchmark prompts or create a log
|
||||
dump that nobody can audit.
|
||||
|
||||
## Test matrix
|
||||
|
||||
Use the real authenticated 7011 Agent UI and the normal `preheret` picker alias.
|
||||
Use `sft_alex_creator` for reversible writes. Never mutate the personal account
|
||||
from an automated test.
|
||||
|
||||
### A. Every one of the ten families
|
||||
|
||||
For calendar, notes, email, tasks, documents, memory, skills, Cookbook/admin,
|
||||
search/browser, and shell/files, test:
|
||||
|
||||
- direct request;
|
||||
- natural misspelling;
|
||||
- ambiguous same-family follow-up;
|
||||
- switch to another family and back;
|
||||
- no-tool greeting before the tool request;
|
||||
- requested count/field limit;
|
||||
- backend failure rendered truthfully;
|
||||
- reload the permalink before a follow-up.
|
||||
|
||||
### B. Stateful mutation families
|
||||
|
||||
For notes, calendar, tasks, documents, memory, and skills:
|
||||
|
||||
- create → verify by API → referential correction → verify;
|
||||
- create → list/read → correction → verify;
|
||||
- typo correction such as name/date/title without repeating the family noun;
|
||||
- correction after one unrelated conversational turn;
|
||||
- destructive request is denied atomically;
|
||||
- failed write never produces a success claim;
|
||||
- cleanup deletes only the UUID-owned test artifact and verifies absence.
|
||||
|
||||
### C. Search and browser conversations
|
||||
|
||||
- search → summarize existing results without a new call;
|
||||
- search → inspect one result with `web_fetch`;
|
||||
- poor results → refine query once;
|
||||
- insufficient evidence → say so without fabrication;
|
||||
- Web toggle combinations `00`, `01`, `10`, and `11` across two turns;
|
||||
- private browser open/snapshot/click only after its permission boundary is
|
||||
deliberately enabled and specified; do not smuggle it in via web search.
|
||||
|
||||
Grade source relevance, freshness, authority, and whether claims are supported,
|
||||
not merely whether `web_search` was called.
|
||||
|
||||
### D. Email and shell
|
||||
|
||||
- Separate fixture accuracy from production connectivity. A fixture pass cannot
|
||||
promote production email health.
|
||||
- Test account listing, inbox listing, reading, and referential follow-up against
|
||||
the configured production-like backend before enabling email actions.
|
||||
- Shell remains toggle-gated. Test off/on transitions, canonical raw command
|
||||
dispatch, read-only output, and denial of network/destructive commands.
|
||||
|
||||
### E. Rendering and performance
|
||||
|
||||
- Assert first visible streamed token, monotonic DOM growth, one final answer,
|
||||
persistence/reload equality, stop behavior, and structured list rendering.
|
||||
- Record request preparation, TTFT, tool duration, post-tool TTFT, total time,
|
||||
input/output tokens, and tool-result bytes.
|
||||
- Diagnose the 30–40 second 7011 restart separately from inference latency.
|
||||
- Bound large calendar/search results before replaying them into later rounds,
|
||||
while preserving IDs and fields needed for follow-ups.
|
||||
|
||||
### F. Image/VL recognition
|
||||
|
||||
- Attach real PNG, JPEG, and WebP images through the 7011 UI and verify the
|
||||
trained model receives native multimodal message content on its clean route.
|
||||
- Test object recognition, visible text/OCR, spatial relationships, charts, and
|
||||
screenshots. Score required facts instead of stylistic wording.
|
||||
- Test image → ambiguous follow-up, image → tool request, and tool result → image
|
||||
comparison without requiring the user to attach the same image again.
|
||||
- Verify image references survive persistence and permalink reload without raw
|
||||
base64, local paths, or hidden wrappers appearing in chat output.
|
||||
- Separate direct model vision from `inspect_media`, browser screenshots, and
|
||||
image generation. The harness must not silently substitute one for another.
|
||||
- Compare the fine-tune with its base VL model on the same images to detect
|
||||
whether tool training regressed visual understanding.
|
||||
|
||||
### G. Regular-model legacy RAG and tool coverage
|
||||
|
||||
- Inventory every enabled non-Odysseus endpoint/model visible in 7011, including
|
||||
its provider, schema mode, native-tool support, context limit, and configured
|
||||
permissions. Do not assume every provider supports the same wire format.
|
||||
- Assert that no non-Odysseus model enters the clean-v3 runtime. These models
|
||||
retain the regular RAG/tool loop and are repaired only in that owning path.
|
||||
- For each model, test every tool family the effective user policy offers:
|
||||
direct request, misspelling, ambiguous follow-up, family switch, backend
|
||||
failure, and Web/Bash toggle transitions. Record unsupported families as an
|
||||
explicit capability limitation, not a silent pass.
|
||||
- Test full schemas versus compact schemas only where both are valid for that
|
||||
model. Store the selected schema mode in every report.
|
||||
- Verify provider-native tool calls, textual fallback parsing where required,
|
||||
canonical argument conversion, execution, evidence replay, and rendering.
|
||||
- Group fixes by shared legacy-runtime or provider-adapter defect. Do not add
|
||||
model-name prompt exceptions when a transport, schema, or RAG ranking issue is
|
||||
responsible.
|
||||
- Maintain a per-model compatibility matrix so adding or changing an endpoint
|
||||
cannot silently regress previously working tools.
|
||||
|
||||
## Fix protocol
|
||||
|
||||
For each failure:
|
||||
|
||||
1. Preserve the raw report and reproduce once on a fresh test session.
|
||||
2. Identify the primary failure layer from the taxonomy above.
|
||||
3. Add the smallest generic red test at that layer.
|
||||
4. Fix the owning module or invariant—not the literal prompt.
|
||||
5. Run the focused unit tests, the original scenario, two adjacent scenarios,
|
||||
and the affected family suite.
|
||||
6. After a batch of category fixes, rerun the ten-family matrix and legacy-RAG
|
||||
isolation test. Do not rerun training unless the contract and harness are
|
||||
proven correct and failures remain model-owned.
|
||||
|
||||
If three failures share a layer, pause case-by-case patching and refactor that
|
||||
layer before continuing.
|
||||
|
||||
## Execution phases
|
||||
|
||||
### Phase 1 — Make the runtime auditable
|
||||
|
||||
- Add a sanitized per-turn decision record: model runtime, contract, proposed
|
||||
calls, policy decisions with reason codes, executions, render owner, timings.
|
||||
- Add startup/runtime provenance to the UI so a linked chat proves which harness
|
||||
handled it.
|
||||
- Add the offered-versus-policy compatibility self-test.
|
||||
- Correct stale preview documentation.
|
||||
|
||||
### Phase 2 — Build the conversation suite
|
||||
|
||||
- Extend the current Playwright verifier with reusable multi-turn scenarios and
|
||||
reversible artifact fixtures.
|
||||
- Implement the matrix above, prioritizing search continuations and all
|
||||
stateful corrections because real usage already exposed those gaps.
|
||||
- Run independent family groups in parallel, but serialize writes that share a
|
||||
backend or fixture account.
|
||||
- Add a small versioned VL fixture set with locally generated, non-private
|
||||
images and deterministic answer keys.
|
||||
|
||||
### Phase 3 — Repair by architecture category
|
||||
|
||||
- Consolidate model-specific runtime selection in one function.
|
||||
- Represent prior successful objects explicitly for referential follow-ups.
|
||||
- Align tool capability classification, contract offering, and policy decisions.
|
||||
- Standardize tool results into bounded envelopes with source/object IDs.
|
||||
- Keep search refinement and evidence sufficiency generic.
|
||||
|
||||
### Phase 4 — Accuracy and speed comparison
|
||||
|
||||
- Compare the clean fine-tune with the base model using identical compact tools,
|
||||
prompts, toggles, backend state, and semantic scoring.
|
||||
- Report functional accuracy, argument accuracy, unsupported success claims,
|
||||
TTFT, total latency, and tokens. Do not compare one model on full schemas and
|
||||
another on compact schemas.
|
||||
- Only consider more SFT/RL for failures classified as model-owned after the
|
||||
harness audit.
|
||||
|
||||
### Phase 4B — Regular-model repair and verification
|
||||
|
||||
- Snapshot the enabled non-Odysseus model inventory.
|
||||
- Run the legacy-RAG compatibility matrix in bounded parallel groups, respecting
|
||||
endpoint rate limits and shared backend write serialization.
|
||||
- Fix shared harness/provider defects first, then rerun all affected models.
|
||||
- Publish separate per-model scores and limitations; do not blend them into the
|
||||
Odysseus fine-tune score.
|
||||
|
||||
### Phase 5 — Ship gate
|
||||
|
||||
Ship only when:
|
||||
|
||||
- every family is at least 90% on sealed functional holdout;
|
||||
- overall functional accuracy is at least 95%;
|
||||
- realistic follow-up suite is at least 95%, with no repeated failure category;
|
||||
- image/VL fixture accuracy does not regress materially from the base model and
|
||||
all attachment/follow-up/persistence flows pass;
|
||||
- routing/execution and safety invariants are 100%;
|
||||
- all reversible writes are API-verified and cleaned up;
|
||||
- search quality and production email are reported separately and honestly;
|
||||
- non-Odysseus models demonstrably retain legacy RAG;
|
||||
- every enabled regular model has a complete tested-tool compatibility record,
|
||||
and every tool advertised as supported passes its functional checks;
|
||||
- no hidden prompt leakage, duplicate rendering, or false success remains;
|
||||
- pre-heretic passing weights and merged adapter backups remain recoverable.
|
||||
|
||||
## Immediate next batch
|
||||
|
||||
1. Expand VL fixtures to charts, screenshots, and image-to-tool turns;
|
||||
investigate the shared base-model OCR limitation without hiding it behind a
|
||||
silent external fallback.
|
||||
2. Add deliberately permissioned private-browser open/snapshot/click checks;
|
||||
keep browser interaction unavailable when its boundary is not enabled.
|
||||
3. Bring the two configured local regular models online and run their matrix.
|
||||
4. Compare fine-tune versus untouched base with identical compact contracts,
|
||||
backend state, prompts, and timing instrumentation.
|
||||
5. Run the sealed all-action holdout and prioritize failures by shared
|
||||
layer rather than by prompt.
|
||||
@@ -0,0 +1,445 @@
|
||||
# Plan: Odysseus Professional Photo Editor
|
||||
|
||||
> Source PRD: Conversation goal, "a Photoshop/Photopea clone with Odysseus style"
|
||||
|
||||
## Product boundary
|
||||
|
||||
Odysseus should provide the editing loop people expect from a professional
|
||||
layer-based photo editor without copying Photoshop's visual design or trying to
|
||||
match every specialist feature. The target is a dependable browser editor for
|
||||
real photo work: direct manipulation, non-destructive layers, precise masking,
|
||||
retouching, typography, export, recovery, and optional AI assistance.
|
||||
|
||||
The existing quiet Odysseus interface remains the visual language. Dense tools
|
||||
are acceptable, but controls should stay restrained, compact, predictable, and
|
||||
usable on both desktop and touch devices.
|
||||
|
||||
## Existing foundation
|
||||
|
||||
The current editor already provides meaningful parts of this product:
|
||||
|
||||
- Raster and editable text layers
|
||||
- Multi-layer selection, nested groups, clipping, visibility, opacity, and locks
|
||||
- Layer, group, and selection masks
|
||||
- Marquee, lasso, wand, SAM, Quick Mask, and saved selections
|
||||
- Brush, eraser, clone, crop, transform, and text tools
|
||||
- Blend modes, adjustment stacks, blur, and several image corrections
|
||||
- Rulers, guides, grid, snapping, zooming, and panning
|
||||
- Undo/redo history with a memory budget
|
||||
- Versioned layered-project serialization, autosave drafts, recovery, and export
|
||||
- Optional endpoint-backed inpaint and image-processing tools
|
||||
- Desktop and mobile editor layouts with Playwright release-gate coverage
|
||||
|
||||
## Architectural decisions
|
||||
|
||||
Durable decisions that apply across every phase:
|
||||
|
||||
- **Editor ownership**: The editor remains an Odysseus feature. Do not embed a
|
||||
third-party editor or imitate another product's chrome.
|
||||
- **Document format**: Continue the versioned Odysseus editor document. Every
|
||||
new persistent capability requires a migration, validation, round-trip test,
|
||||
and corrupt-input recovery behavior.
|
||||
- **Layer model**: Grow the document into explicit layer kinds rather than
|
||||
hiding more behavior in raster canvases. The intended kinds are raster, text,
|
||||
shape, adjustment, and placed/smart content.
|
||||
- **Non-destructive default**: Preserve source pixels and editable parameters
|
||||
whenever practical. Destructive actions remain available as explicit Apply,
|
||||
Rasterize, or Merge commands.
|
||||
- **Interaction engine**: Transform, crop, selections, text frames, masks, and
|
||||
shapes share one pointer-session model for hit testing, pointer capture,
|
||||
modifiers, snapping, cancellation, and undo transactions.
|
||||
- **Rendering**: Keep Canvas 2D as the compatibility renderer initially. Move
|
||||
expensive compositing and pixel operations behind renderer/worker boundaries
|
||||
before considering WebGL or WebGPU acceleration.
|
||||
- **History**: One continuous gesture creates one undo entry. Preview frames are
|
||||
never separate history entries, and Cancel restores the exact starting state.
|
||||
- **Persistence routes**: Continue using `/api/editor-drafts` for layered draft
|
||||
persistence and `/api/gallery` for media-library save/replace operations.
|
||||
- **AI boundary**: AI features consume capability-based image endpoints. Core
|
||||
editing never requires a particular model, repository, or provider.
|
||||
- **Responsive behavior**: Desktop favors precision; touch targets gain larger
|
||||
invisible hit areas without visually enlarging the whole interface.
|
||||
- **Testing**: Every phase adds deterministic geometry/unit tests and at least
|
||||
one complete Playwright workflow covering persistence and undo where relevant.
|
||||
- **Incremental architecture**: New behavior leaves the main editor orchestrator
|
||||
through small domain modules. Avoid broad refactors that do not deliver a
|
||||
visible editing improvement in the same phase.
|
||||
|
||||
---
|
||||
|
||||
## Phase 1: Accurate Transform Frame
|
||||
|
||||
**User stories**: I can clearly see and grab the transform frame at any zoom. I
|
||||
can resize from corners or sides without grabbing invisible or incorrect areas.
|
||||
|
||||
### What to build
|
||||
|
||||
Replace the four-corner-only frame with a shared frame geometry model. Render
|
||||
four corners, four edge handles, a rotation control, and an optional center
|
||||
pivot from the same geometry used for hit testing. Keep handles visually compact
|
||||
while providing touch-sized invisible targets. Make the frame stay aligned
|
||||
during zoom, pan, viewport resize, and when handles extend outside the image.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Eight resize handles, rotation control, and center pivot derive from one geometry result.
|
||||
- [x] Drawn handles and hit targets cannot disagree.
|
||||
- [x] Handles remain a stable visual size from minimum to maximum zoom.
|
||||
- [x] Touch hit targets are at least 40 CSS pixels without oversized visuals.
|
||||
- [x] Outside-canvas handles remain interactive and visible when space permits.
|
||||
- [x] Hover and active cursors match each handle's current screen direction.
|
||||
- [x] Desktop and mobile Playwright tests grab every handle successfully.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2: Correct Rotated Resize
|
||||
|
||||
**User stories**: I can resize a rotated layer naturally. The opposite side or
|
||||
corner stays fixed, and the frame follows my pointer rather than drifting.
|
||||
|
||||
### What to build
|
||||
|
||||
Calculate drag movement in the frame's rotated local coordinate system. Anchor
|
||||
the opposite handle in document space and derive the new center from that
|
||||
anchor. Support crossing an axis as a deliberate flip instead of clamping to a
|
||||
one-pixel box. Apply the same geometry to one layer, multiple layers, and a
|
||||
selection transform.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Rotated corner and edge drags follow the pointer on the frame's local axes.
|
||||
- [x] The opposite anchor remains fixed within a sub-pixel tolerance.
|
||||
- [x] Crossing width or height zero produces a predictable horizontal or vertical flip.
|
||||
- [x] Shift locks the starting aspect ratio.
|
||||
- [x] Alt/Option scales around the transform center.
|
||||
- [x] Combined Shift+Alt/Option behavior is deterministic.
|
||||
- [x] Rotation snaps to 15-degree increments with Shift and remains smooth otherwise.
|
||||
- [x] Geometry tests cover 0, 45, 90, 135, and arbitrary-degree rotations.
|
||||
|
||||
---
|
||||
|
||||
## Phase 3: Transform Interaction Polish
|
||||
|
||||
**User stories**: Transform behaves like a professional tool on mouse, pen, and
|
||||
touch. I can see exact values, snap precisely, and never lose a drag at the edge.
|
||||
|
||||
### What to build
|
||||
|
||||
Use a unified pointer session with pointer capture, live modifiers, and a small
|
||||
contextual transform readout. Add accurate rotated-frame interior hit testing,
|
||||
keyboard nudging, frame snapping, and clear Apply/Cancel behavior. Keep the
|
||||
existing compact Odysseus styling and make the numeric popup a precision surface
|
||||
rather than a competing transform implementation.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Pointer capture keeps a drag alive outside the canvas and browser viewport.
|
||||
- [x] Clicking inside a rotated frame moves it; clicking its empty bounding-box corner does not.
|
||||
- [x] Live X, Y, W, H, and angle values stay synchronized with direct manipulation.
|
||||
- [x] Arrow keys nudge, Shift+Arrow performs a larger nudge, Enter applies, and Escape cancels.
|
||||
- [x] Layer edges, document center/edges, guides, and grid participate in transform snapping.
|
||||
- [x] Snap guides clearly identify the active alignment without obscuring the photo.
|
||||
- [x] A complete gesture creates exactly one undo step.
|
||||
- [x] Touch gestures do not conflict with viewport pinch/pan behavior.
|
||||
|
||||
---
|
||||
|
||||
## Phase 4: Transform Content Correctness
|
||||
|
||||
**User stories**: Transforming layers never unexpectedly damages masks, text,
|
||||
group layout, clipping, or image quality. Saving and reopening preserves it.
|
||||
|
||||
### What to build
|
||||
|
||||
Route raster layers, text layers, linked and unlinked masks, selections, clipped
|
||||
layers, and grouped multi-selection through the same transform contract. Keep
|
||||
immutable source data during previews and validate the final result through
|
||||
undo, cancel, autosave, project download, and reopen.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Raster previews are always derived from the session source, never a prior preview.
|
||||
- [x] Editable text remains editable after scaling, rotation, and flipping.
|
||||
- [x] Linked masks follow the layer while unlinked masks remain in document space.
|
||||
- [x] Multi-layer transforms preserve relative centers, order, clipping, and group membership.
|
||||
- [x] Transforming a selection changes only the selection mask unless content transform is explicitly chosen.
|
||||
- [x] Apply, Cancel, Undo, Redo, autosave reopen, and project-file reopen produce matching pixels and metadata.
|
||||
- [x] Large transforms cannot allocate beyond the editor's documented surface budget.
|
||||
|
||||
---
|
||||
|
||||
## Phase 5: Shared Direct-Manipulation Sessions
|
||||
|
||||
**User stories**: Crop, selections, masks, text boxes, and shapes feel consistent
|
||||
with Transform instead of each behaving like a separate mini application.
|
||||
|
||||
### What to build
|
||||
|
||||
Generalize the proven transform pointer session into a reusable interaction
|
||||
contract. Migrate crop and selection movement first as a visible tracer bullet,
|
||||
including modifiers, snapping, pointer capture, cancel, and one-step history.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Transform, crop, and selection movement use the same gesture lifecycle.
|
||||
- [x] Tool switching safely commits, cancels, or prompts according to one policy.
|
||||
- [x] No stale pointer session can modify a newly selected tool or document.
|
||||
- [x] Mouse, pen, and touch event behavior is covered by shared tests.
|
||||
- [x] Adding a future frame-based tool does not require another global event stack.
|
||||
|
||||
---
|
||||
|
||||
## Phase 6: Non-Destructive Placed Layers
|
||||
|
||||
**User stories**: I can import an image, resize it repeatedly without cumulative
|
||||
quality loss, replace its source, and choose when to rasterize it.
|
||||
|
||||
### What to build
|
||||
|
||||
Introduce a placed/smart layer kind containing source pixels and persistent
|
||||
transform metadata. Import-as-layer uses this kind by default. Rendering applies
|
||||
the transform at composite time, while Rasterize produces a normal raster layer.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Repeated transforms render from the original source rather than resampling the last result.
|
||||
- [x] A placed layer can be replaced while preserving its transform and masks.
|
||||
- [x] Rasterize produces a visually matching editable raster layer.
|
||||
- [x] Masks, clipping, groups, blend modes, and opacity work with placed layers.
|
||||
- [x] Version migration and recovery handle missing or corrupt placed sources.
|
||||
- [x] Existing raster projects open without changed output.
|
||||
|
||||
---
|
||||
|
||||
## Phase 7: Professional Selections And Masks
|
||||
|
||||
**User stories**: I can build, inspect, refine, save, transform, and reuse precise
|
||||
selections without manually repainting every edge.
|
||||
|
||||
### What to build
|
||||
|
||||
Unify marquee, lasso, wand, SAM, Quick Mask, and saved selections around one
|
||||
selection-mask model. Add explicit replace/add/subtract/intersect modes, feather,
|
||||
expand, contract, smooth, border, and a focused refine-edge workflow.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Every selection tool supports replace, add, subtract, and intersect modes.
|
||||
- [x] Feather, expand, contract, smooth, and border preview before applying.
|
||||
- [x] Quick Mask edits the same canonical selection shown by marching ants.
|
||||
- [x] Selection-to-layer-mask and layer-mask-to-selection round-trip accurately.
|
||||
- [x] Saved selections retain names and pixels across reopen.
|
||||
- [x] Edge refinement works without requiring an AI dependency.
|
||||
|
||||
---
|
||||
|
||||
## Phase 8: Paint And Retouch Workflow
|
||||
|
||||
**User stories**: I can paint and retouch photographs with predictable strokes,
|
||||
reusable presets, and the controls expected for a mouse, pen, or touch device.
|
||||
|
||||
### What to build
|
||||
|
||||
Promote brush behavior into a reusable brush engine. Add spacing, smoothing,
|
||||
pressure mapping, blend mode, sampled color, presets, and stroke preview. Build
|
||||
healing, dodge, and burn as complete retouching paths using that engine.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Brush, eraser, clone, masks, and inpaint share spacing and smoothing behavior.
|
||||
- [x] Pressure can independently affect size, opacity, or flow when supported.
|
||||
- [x] Eyedropper samples composite or active-layer color.
|
||||
- [x] Brush presets can be created, named, selected, and deleted.
|
||||
- [x] Healing, dodge, and burn create one undo entry per stroke.
|
||||
- [x] Long strokes remain smooth without blocking the main interface.
|
||||
|
||||
---
|
||||
|
||||
## Phase 9: Editable Text And Shapes
|
||||
|
||||
**User stories**: I can design labels, cards, and overlays with text and vector
|
||||
shapes that remain editable after saving and reopening.
|
||||
|
||||
### What to build
|
||||
|
||||
Add on-canvas text-frame editing, selection, caret behavior, typography, and
|
||||
alignment. Introduce shape layers for rectangle, ellipse, line, and path-backed
|
||||
polygons with editable fill, stroke, corners, and transform metadata.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [x] Text is edited directly on canvas without immediately rasterizing.
|
||||
- [x] Font, size, weight, line height, letter spacing, alignment, and color persist.
|
||||
- [x] Rectangle, ellipse, line, and polygon shapes remain editable.
|
||||
- [x] Shape fill, stroke, width, and corner radius can be changed after creation.
|
||||
- [x] Text and shape layers support masks, clipping, groups, blend modes, and transform.
|
||||
- [x] Missing fonts fall back predictably without corrupting the project.
|
||||
|
||||
---
|
||||
|
||||
## Phase 10: Adjustment Layers And Color
|
||||
|
||||
**User stories**: I can correct a photograph non-destructively and return later
|
||||
to modify the correction without reconstructing the edit.
|
||||
|
||||
### What to build
|
||||
|
||||
Promote adjustments into first-class layers with masks and clipping. Deliver
|
||||
Levels and Curves first, then exposure, white balance, hue/saturation, color
|
||||
balance, selective color, gradients, and channel-aware controls.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] Adjustment layers affect content below them and can be clipped or grouped.
|
||||
- [ ] Every adjustment has live preview, reset, visibility, opacity, mask, Apply, and Cancel behavior.
|
||||
- [ ] Levels includes histogram, input range, gamma, and output range.
|
||||
- [ ] Curves supports RGB and channel curves with editable points.
|
||||
- [ ] Color results match flattened export and project reopen.
|
||||
- [ ] Large previews are throttled or worker-backed and remain cancellable.
|
||||
|
||||
---
|
||||
|
||||
## Phase 11: Layer Effects And Filters
|
||||
|
||||
**User stories**: I can add common visual effects without permanently altering
|
||||
the layer and can reorder or disable those effects later.
|
||||
|
||||
### What to build
|
||||
|
||||
Create an ordered non-destructive filter/effect stack. Begin with Gaussian blur,
|
||||
sharpen, shadow, stroke, and color overlay; then add filter masks and reusable
|
||||
effect presets.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] Effects can be added, reordered, toggled, edited, masked, and removed.
|
||||
- [ ] Drop shadow, stroke, color overlay, blur, and sharpen survive project reopen.
|
||||
- [ ] Effects render correctly inside groups and clipping stacks.
|
||||
- [ ] Apply/rasterize produces a pixel-equivalent raster result.
|
||||
- [ ] Expensive filters expose progress and cancellation.
|
||||
|
||||
---
|
||||
|
||||
## Phase 12: Odysseus Professional Workspace
|
||||
|
||||
**User stories**: I can work quickly without fighting floating windows or losing
|
||||
the active tool, layer, selection, or document context.
|
||||
|
||||
### What to build
|
||||
|
||||
Refine the existing shell into a consistent professional workspace: contextual
|
||||
tool options, properties inspector, panel persistence, command search, status
|
||||
information, multi-document switching, and compact touch sheets. Preserve the
|
||||
current Odysseus palette, typography, restrained borders, and frosted surfaces.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] Tool options appear in one predictable location and never duplicate popup state.
|
||||
- [ ] Panels remember size, collapsed state, and position per device class.
|
||||
- [ ] The properties inspector follows the active layer, mask, selection, or tool.
|
||||
- [ ] Command search exposes actions and shortcuts without adding toolbar clutter.
|
||||
- [ ] Switching documents preserves independent history, zoom, pan, and selection.
|
||||
- [ ] Mobile prioritizes canvas area while keeping all commands reachable.
|
||||
|
||||
---
|
||||
|
||||
## Phase 13: File Interchange And Export
|
||||
|
||||
**User stories**: I can bring common assets into Odysseus and export predictable
|
||||
results without losing transparency, dimensions, or color intent.
|
||||
|
||||
### What to build
|
||||
|
||||
Strengthen image import/export first, then add layered interchange where a
|
||||
maintained parser makes it safe. Keep Odysseus project files as the lossless
|
||||
source of truth and clearly report what an external format cannot preserve.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] PNG, JPEG, WebP, and supported modern image imports honor orientation and transparency.
|
||||
- [ ] Export exposes format, dimensions, quality, metadata, and transparency choices.
|
||||
- [ ] Copy/paste and drag/drop preserve alpha and use placed layers when appropriate.
|
||||
- [ ] Layered imports report unsupported features instead of silently flattening them.
|
||||
- [ ] Exported pixels are covered by deterministic visual comparisons.
|
||||
|
||||
---
|
||||
|
||||
## Phase 14: Large-Document Performance And Recovery
|
||||
|
||||
**User stories**: Large photos and layered projects remain responsive, autosave
|
||||
reliably, and recover after a crash or interrupted network connection.
|
||||
|
||||
### What to build
|
||||
|
||||
Move serialization, thumbnails, filters, and suitable pixel operations into
|
||||
workers. Add dirty-region rendering, reusable surfaces, measurable memory
|
||||
budgets, operation cancellation, autosave generations, and recovery diagnostics.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] Normal interactions remain responsive on the agreed 4K multi-layer benchmark.
|
||||
- [ ] Compositing avoids rebuilding unaffected layers and thumbnails.
|
||||
- [ ] History and document surfaces stay within explicit memory limits.
|
||||
- [ ] Closing or switching documents cancels stale work safely.
|
||||
- [ ] Autosave never lets an older request overwrite newer state.
|
||||
- [ ] Recovery can identify the last complete generation and explain skipped data.
|
||||
|
||||
---
|
||||
|
||||
## Phase 15: Odysseus-Native Assisted Editing
|
||||
|
||||
**User stories**: I can use an available local or remote image capability as an
|
||||
editing assistant while retaining masks, layers, undo, privacy choices, and
|
||||
normal manual controls.
|
||||
|
||||
### What to build
|
||||
|
||||
Standardize image capability discovery and requests for generation, editing,
|
||||
inpainting, segmentation, restoration, and upscaling. Results enter the document
|
||||
as named layers with provenance and reusable masks. Add orchestration only after
|
||||
the manual operation it assists is dependable.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] The UI describes required capabilities rather than model or provider names.
|
||||
- [ ] Memory and unrelated chat context are not sent to image endpoints.
|
||||
- [ ] Requests show progress, support cancellation, and cannot update a closed document.
|
||||
- [ ] Generated results arrive as reversible layers with prompt/settings metadata.
|
||||
- [ ] A failed endpoint leaves the source document unchanged and offers a useful retry path.
|
||||
- [ ] Manual selection and masking remain available when assisted tools are absent.
|
||||
|
||||
---
|
||||
|
||||
## Phase 16: Professional Release Gate
|
||||
|
||||
**User stories**: I can trust the editor for real work and understand what is
|
||||
unsupported before committing an edit.
|
||||
|
||||
### What to build
|
||||
|
||||
Create a release gate around complete user journeys rather than isolated button
|
||||
tests. Cover accessibility, keyboard-only operation, touch, browser differences,
|
||||
pixel correctness, persistence, failure recovery, and large-document behavior.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
- [ ] Core workflows pass on current Chromium and Firefox desktop builds.
|
||||
- [ ] Mobile workflows pass at representative phone and tablet viewports.
|
||||
- [ ] Keyboard-only users can reach every command and escape every modal state.
|
||||
- [ ] Transform, masks, text, adjustments, export, and reopen have pixel/metadata regression tests.
|
||||
- [ ] No supported action silently flattens or discards editable document data.
|
||||
- [ ] The ALPHA badge can be removed based on explicit reliability metrics.
|
||||
|
||||
---
|
||||
|
||||
## Recommended delivery order
|
||||
|
||||
The first four phases are one focused Transform 2.0 program and should ship in
|
||||
order. Phases 5 and 6 establish the interaction and document foundations needed
|
||||
for the remaining professional tools. After that, phases 7 through 13 can be
|
||||
prioritized by user value, while performance and release-gate work continue as
|
||||
part of every phase rather than being deferred entirely to the end.
|
||||
|
||||
The recommended first milestone is complete when Phases 1 through 4 are live:
|
||||
transforming one layer, multiple layers, text, masks, and selections feels
|
||||
precise on desktop and mobile and remains correct through undo and reopen.
|
||||
@@ -0,0 +1,159 @@
|
||||
# Photo Editor Remaining Scope
|
||||
|
||||
Date: 2026-08-29
|
||||
|
||||
## Current verdict
|
||||
|
||||
Odysseus is now a credible layered everyday editor, not an editor mockup. The
|
||||
first nine roadmap phases are implemented: professional transform geometry,
|
||||
shared direct-manipulation sessions, retained placed content, unified
|
||||
selections and masks, a reusable brush/retouch engine, and retained text and
|
||||
shape layers.
|
||||
|
||||
Phase 10 is functionally advanced but not closed. First-class adjustment layers
|
||||
now support Levels, Curves, Exposure, White Balance, Brightness/Contrast,
|
||||
Hue/Saturation/Lightness, Color Balance, Selective Color, and Gradient Map.
|
||||
They participate in clipping, groups, masks, visibility, opacity, history, the
|
||||
v14 document format, and flattening. Retained effects have since been added as
|
||||
a separate ordered stack with Gaussian Blur, Color Overlay, Drop Shadow, and
|
||||
Stroke, including editable colors, visibility, opacity, reorder, rasterize,
|
||||
history, persistence, and migration.
|
||||
|
||||
Practical readiness estimate:
|
||||
|
||||
- Everyday layered photo editing: **about 88%**
|
||||
- Dependable professional v1 described by the roadmap: **about 62%**
|
||||
- Broad Photoshop/Photopea feature parity: **about 50%**
|
||||
|
||||
The remaining gap is dominated by large-document rendering outside the live
|
||||
composite path, workspace consolidation, interchange/color policy, and release
|
||||
proof rather than basic canvas tools.
|
||||
|
||||
## Verification snapshot
|
||||
|
||||
- The focused editor unit suite currently passes **31 tests** in Docker.
|
||||
- The full photo-editor browser suite currently has **41 passing workflows**;
|
||||
the nested-group selection workflow initially exposed a row-hit regression,
|
||||
which now passes on isolated rerun after the slider-selection fix. The new
|
||||
group-effects workflow also passes.
|
||||
- The new adjustment tests exercise deterministic pixel math, nested parameter
|
||||
normalization, retained metadata, undo/redo, clipping, masks, and draft
|
||||
reopen.
|
||||
- The latest editor changes have not yet been rebuilt into the live `7011`
|
||||
container.
|
||||
|
||||
## Close Phase 10
|
||||
|
||||
This is the immediate release slice.
|
||||
|
||||
1. Finish the bounded preview path for large documents. Downsampled previews
|
||||
now keep control movement responsive and full resolution is restored for
|
||||
commit/export. Live worker composites now use generation checks, latest-only
|
||||
coalescing, and close/reopen invalidation; extend the same guarantees to
|
||||
remaining preview paths.
|
||||
2. Add flattened-export versus reopened-project pixel comparisons for every
|
||||
adjustment family, including groups, clipping, masks, blend mode, and
|
||||
partial opacity.
|
||||
3. Validate the color algorithms visually. White Balance and Selective Color
|
||||
are currently deterministic approximations, not color-managed photographic
|
||||
transforms.
|
||||
4. Test every adjustment popup on phone and desktop viewports, including tall
|
||||
popups, color inputs, drag, Reset, Apply, Cancel, and Escape.
|
||||
5. Decide the migration path for the older per-raster `adjLayers` stack. It can
|
||||
remain readable for compatibility, but new UI should converge on first-class
|
||||
adjustment layers instead of maintaining two competing concepts.
|
||||
6. Bump static cache versions, rebuild the live container, and run a short
|
||||
visual smoke test on `7011`.
|
||||
|
||||
## Phase 11: Retained effects and filters
|
||||
|
||||
The retained-effects slice is implemented for raster/placed/text/shape-compatible
|
||||
layer output: Gaussian Blur, Sharpen, Color Overlay, Drop Shadow, and Stroke
|
||||
have editable colors/parameters, visibility, opacity, reorder, rasterize,
|
||||
history, migration, and reopen support. Effect-specific masks, presets, and
|
||||
group-level effects are also implemented and covered by focused browser tests.
|
||||
Remaining work is:
|
||||
|
||||
1. Extend worker coverage to serialization and remaining preview paths.
|
||||
Thumbnail encoding, retained-effect rasterization, and live composite
|
||||
rendering now use a worker where OffscreenCanvas is available, with
|
||||
synchronous compatibility fallbacks. Generation invalidation, latest-only
|
||||
coalescing, and CPU loop cancellation protect live rendering.
|
||||
2. Add explicit group-effect blend/ordering tests for nested groups and
|
||||
non-default blend modes, plus visual comparisons for effect stacks.
|
||||
|
||||
Introduce the renderer/worker cancellation boundary here rather than adding
|
||||
more synchronous full-canvas filters that Phase 14 must immediately replace.
|
||||
|
||||
## Phase 12: Professional workspace
|
||||
|
||||
Consolidate fragmented popups into one contextual properties surface. Persist
|
||||
panel layout by device class, add command search, expose stable document status,
|
||||
and support multiple open documents with independent history, zoom, pan, and
|
||||
selection. Mobile should use canvas-first sheets rather than compressed desktop
|
||||
panels.
|
||||
|
||||
## Phase 13: Interchange and export
|
||||
|
||||
Harden orientation, transparency, metadata, and color behavior for PNG, JPEG,
|
||||
and WebP first. Add copy/paste and drag/drop through placed layers. Treat
|
||||
layered formats as explicit compatibility projects: unsupported PSD/TIFF/HEIC
|
||||
features must be reported, never silently discarded. Odysseus project files
|
||||
remain the lossless source of truth.
|
||||
|
||||
## Phase 14: Performance and recovery
|
||||
|
||||
Move remaining preview/pixel paths into workers. Thumbnail encoding,
|
||||
autosave serialization, adjustment rendering, and retained-effect rendering
|
||||
now have worker-backed paths with compatibility fallbacks. Add
|
||||
dirty-region compositing, reusable render surfaces, cancellation tokens,
|
||||
operation telemetry, a documented surface/history budget, autosave generations,
|
||||
and a checked-in 4K multi-layer benchmark.
|
||||
|
||||
This phase is the main architectural risk. Canvas 2D remains a valid
|
||||
compatibility renderer, but full-document synchronous passes will not scale to
|
||||
professional documents.
|
||||
|
||||
## Phase 15: Assisted editing
|
||||
|
||||
Normalize generation, editing, inpainting, segmentation, restoration, and
|
||||
upscaling behind capability-based endpoints. Keep model/provider names out of
|
||||
editor logic. Requests must exclude chat memory, show progress, cancel safely,
|
||||
and return named reversible layers with provenance. Manual tools remain fully
|
||||
usable without an endpoint.
|
||||
|
||||
Much of the endpoint plumbing already exists; the remaining work is consistent
|
||||
capability discovery, lifecycle safety, and editor-native result handling.
|
||||
|
||||
## Phase 16: Release gate
|
||||
|
||||
Run complete user journeys on Chromium and Firefox desktop plus representative
|
||||
phone/tablet viewports. Add keyboard-only and accessibility coverage, mixed
|
||||
20-edit persistence/export tests, failure recovery, and large-document stress
|
||||
tests. No supported operation may silently flatten or discard retained state.
|
||||
|
||||
## Architecture debt to control
|
||||
|
||||
- `galleryEditor.js` is still a large orchestrator. Continue extracting domain
|
||||
modules as visible features move, without a broad rewrite.
|
||||
- Legacy raster adjustment sublayers and first-class adjustment layers overlap.
|
||||
Converge on the first-class model.
|
||||
- Pixel effects still rely heavily on synchronous full-canvas work.
|
||||
- `static/style.css` carries substantial editor-specific surface area and needs
|
||||
clearer component boundaries before workspace customization expands.
|
||||
- The repository worktree contains many unrelated changes. Editor release and
|
||||
merge decisions require a scoped diff or clean integration branch.
|
||||
|
||||
## Recommended execution order
|
||||
|
||||
1. Close and deploy Phase 10.
|
||||
2. Build Phase 11 through a cancellable render boundary.
|
||||
3. Consolidate the workspace in Phase 12.
|
||||
4. Define color/metadata policy and complete Phase 13.
|
||||
5. Finish worker rendering, stress, and recovery in Phase 14.
|
||||
6. Normalize assisted editing in Phase 15.
|
||||
7. Run the cross-browser professional release gate in Phase 16.
|
||||
|
||||
Do not expand into full PSD fidelity, CMYK production, RAW development, 3D, or
|
||||
complete Photoshop parity before this critical path passes. Those are separate
|
||||
product decisions, not prerequisites for a strong Odysseus editor.
|
||||
@@ -1,4 +1,7 @@
|
||||
# Optional dependencies — install only if you use the corresponding feature.
|
||||
# Local OCR for screenshots, scans, labels, and coordinate-grounded text extraction.
|
||||
rapidocr==3.9.2
|
||||
onnxruntime>=1.20,<2
|
||||
# The app handles their absence gracefully (clear error message on first use).
|
||||
#
|
||||
# Note: chromadb-client + fastembed moved to requirements.txt — RAG, semantic
|
||||
@@ -12,6 +15,16 @@
|
||||
# GPU-accelerated transcription — it's auto-detected, CPU is used otherwise.
|
||||
faster-whisper
|
||||
|
||||
# Local text-to-speech via Kokoro-82M for the "local" TTS provider.
|
||||
# Kokoro 0.9.4 declares Python >=3.10,<3.13; Odysseus itself requires 3.11+,
|
||||
# so pip installs these extras on 3.11-3.12 and deliberately skips them on
|
||||
# Python 3.13+ (including the Python 3.14 container image). Kokoro declares
|
||||
# torch; the local provider still
|
||||
# requires a CUDA-enabled torch build and GPU at runtime. SoundFile is separate
|
||||
# in Kokoro's official install instructions and is not a transitive dependency.
|
||||
kokoro==0.9.4; python_version >= "3.11" and python_version < "3.13"
|
||||
soundfile; python_version >= "3.11" and python_version < "3.13"
|
||||
|
||||
# DuckDuckGo as a search provider option.
|
||||
# Install if you want DDG in the search-provider dropdown.
|
||||
# Alternatives: SearXNG, Brave, Tavily, Serper, Google PSE.
|
||||
@@ -34,3 +47,6 @@ PyMuPDF
|
||||
# [all]/Azure/audio extras (cloud + heavy). Pinned to a release >30 days old per
|
||||
# the dependency-age discussion in issue #485.
|
||||
markitdown[docx,pptx,xlsx,xls]==0.1.6
|
||||
|
||||
# Photoshop PSD opening / flattened previews / layer inspection.
|
||||
psd-tools
|
||||
|
||||
+11
-1
@@ -3,10 +3,16 @@ uvicorn
|
||||
python-multipart
|
||||
python-dotenv
|
||||
httpx
|
||||
httpcore>=1.0,<2.0
|
||||
pydantic>=2.13.4
|
||||
pydantic-settings>=2.14.1
|
||||
SQLAlchemy
|
||||
pypdf
|
||||
pypdfium2
|
||||
Pillow
|
||||
faster-whisper
|
||||
PyPDF2
|
||||
pdfplumber
|
||||
beautifulsoup4
|
||||
charset-normalizer
|
||||
numpy
|
||||
@@ -18,6 +24,7 @@ numpy
|
||||
chromadb-client
|
||||
fastembed
|
||||
youtube-transcript-api
|
||||
yt-dlp
|
||||
# Markdown rendering for research reports (src/visual_report.py).
|
||||
# Imported at module-top so it's a hard core dep, not optional.
|
||||
markdown
|
||||
@@ -37,7 +44,10 @@ python-dateutil
|
||||
caldav
|
||||
cryptography
|
||||
bcrypt
|
||||
mcp
|
||||
# Built-in servers use the v1 low-level Server decorator API. MCP SDK v2 is a
|
||||
# breaking rewrite, so keep fresh installs on the maintained v1 line until the
|
||||
# servers are migrated together.
|
||||
mcp<2
|
||||
pyotp
|
||||
qrcode[pil]
|
||||
croniter
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: artifact-completion
|
||||
description: Create requested artifacts early, iterate from concrete output, and verify final deliverables
|
||||
version: 1.0.0
|
||||
category: agent
|
||||
tags: [artifacts, files, verification, workflow]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when the task requires a file, patch, report, document, image, archive, configuration, or other persistent deliverable rather than only a text answer.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Extract the required deliverable path, format, content constraints, and acceptance criteria.
|
||||
2. Inspect the source material and existing target without delaying the first valid artifact.
|
||||
3. Create a minimal complete version at the required location, then iterate from that concrete output.
|
||||
4. Use the format's native parser, renderer, compiler, or test tool to inspect the artifact.
|
||||
5. Repair specific validation, content, or presentation failures while preserving correct portions.
|
||||
6. Confirm the final path, file type, required content, and usability before reporting completion.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not spend the full task budget inspecting without creating the requested output.
|
||||
- Do not place the artifact at a convenient path when the task specifies another location.
|
||||
- Do not use a filename extension as proof that the file is valid in that format.
|
||||
- Do not report completion while placeholders, missing sections, parse errors, or failed checks remain.
|
||||
|
||||
## Verification
|
||||
|
||||
- The artifact exists at the required path and opens or parses successfully.
|
||||
- Required sections, fields, labels, or visual elements are present.
|
||||
- Relevant tests, render checks, or validators pass.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: terminal-recovery
|
||||
description: Recover from failed terminal commands using evidence-driven diagnosis and bounded retries
|
||||
version: 1.0.0
|
||||
category: agent
|
||||
tags: [terminal, shell, debugging, recovery]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when a command fails, times out, produces incomplete output, or behaves differently from what the task requires.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Read the command, exit status, standard output, and standard error before choosing a response.
|
||||
2. Confirm the working directory, relevant files, executable availability, permissions, and environment assumptions with minimal read-only probes.
|
||||
3. Classify the failure as syntax, missing dependency, wrong path, permissions, resource pressure, timeout, service state, or task logic.
|
||||
4. Change one relevant condition and retry the narrowest command that can test the diagnosis.
|
||||
5. For a long-running command, use the returned session identifier to poll or provide input instead of launching duplicates.
|
||||
6. After recovery, run the original acceptance check and inspect the resulting files or service state.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not rerun an unchanged failing command repeatedly.
|
||||
- Do not install packages or change global configuration before confirming they are missing and necessary.
|
||||
- Do not launch a second server or training job before checking for an existing process and port or device conflicts.
|
||||
- Do not treat partial output or a zero exit status as proof that the requested state was produced.
|
||||
|
||||
## Verification
|
||||
|
||||
- The diagnosed cause is supported by command output or environment state.
|
||||
- The corrected command exits as expected.
|
||||
- The requested artifact, process, or state passes an independent acceptance check.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: tool-discovery
|
||||
description: Discover the smallest capable tool set and confirm argument schemas before acting
|
||||
version: 1.0.0
|
||||
category: agent
|
||||
tags: [tools, discovery, routing, schemas]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when a task requires tools whose names, capabilities, or argument shapes are not already clear. This is especially useful when many tools are available or a previous call failed because the wrong tool or parameters were selected.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Translate the request into required capabilities such as reading, searching, editing, executing, browsing, or verifying.
|
||||
2. Search the tool index for those capabilities and inspect the returned tool descriptions and schemas.
|
||||
3. Prefer one direct tool over a chain of indirect tools when it can complete the operation and provide evidence.
|
||||
4. Check required parameters, identifiers, path rules, side effects, and approval requirements before calling the tool.
|
||||
5. Make a small read-only probe when the environment or target is uncertain.
|
||||
6. Execute the selected action, inspect the result, and only broaden the tool search if the result shows a concrete capability gap.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not guess tool names or argument keys from memory when the index or schema is available.
|
||||
- Do not load unrelated tool groups into context.
|
||||
- Do not repeat the same failed call without changing the arguments or strategy.
|
||||
- Do not use a broad shell or browser workaround when a scoped native tool already owns the operation.
|
||||
|
||||
## Verification
|
||||
|
||||
- The chosen tool directly matches the required capability.
|
||||
- Required arguments follow the exposed schema.
|
||||
- The result contains evidence of the requested effect or a specific error that guides the next step.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: verified-state-change
|
||||
description: Make scoped state changes with target confirmation, minimal mutation, and read-back verification
|
||||
version: 1.0.0
|
||||
category: agent
|
||||
tags: [state, mutation, verification, safety]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when creating, editing, deleting, moving, sending, scheduling, or otherwise changing persistent state through an application, API, filesystem, or service.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Read the current state and identify the target using stable identifiers plus enough content to disambiguate it.
|
||||
2. Preserve fields the user did not ask to change and choose the narrowest supported mutation.
|
||||
3. For destructive or externally visible actions, confirm that the user's instruction authorizes the exact target and effect.
|
||||
4. Perform the mutation once and capture the returned identifier, status, or revision.
|
||||
5. Read the target again through an independent list, fetch, status, or content operation.
|
||||
6. Compare the observed state with the requested outcome and repair only the specific mismatch.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not infer the target from a stale active item when a stable identifier can be fetched.
|
||||
- Do not report success from an accepted request alone; asynchronous or partial operations may not have completed.
|
||||
- Do not replace an entire object when a field-level update is supported and safer.
|
||||
- Do not silently broaden a mutation to adjacent files, records, accounts, or services.
|
||||
|
||||
## Verification
|
||||
|
||||
- The target identity was confirmed before mutation.
|
||||
- A read-back shows the intended values and preserves unrelated state.
|
||||
- Any external effect has a concrete status, identifier, or observable result.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: action-evidence-synthesis
|
||||
description: "Turn messages, meeting notes, and documents into sourced decisions, actions, dependencies, and risks"
|
||||
version: 1.0.0
|
||||
category: communication
|
||||
tags: [messages, meetings, actions, status, evidence]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when information is fragmented across messages, meeting notes, transcripts, or documents and the user needs an action list, status summary, feasibility assessment, or executive brief.
|
||||
|
||||
Do not use when the source material is unavailable or when the user only wants a verbatim transcript.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Identify the requested scope, audience, time window, and decision to support.
|
||||
2. Gather the relevant records in full and preserve stable source identifiers, authors, and timestamps.
|
||||
3. Extract explicit decisions, commitments, requests, owners, dates, dependencies, blockers, and changed facts.
|
||||
4. Reconcile revisions by preferring the newest authoritative record; keep unresolved conflicts visible instead of guessing.
|
||||
5. Separate observed facts from inferred owners, dates, urgency, feasibility, or recommendations, and label every inference as tentative.
|
||||
6. Produce the requested format with concise source references beside consequential claims and a final list of open questions.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not turn discussion or speculation into a confirmed decision.
|
||||
- Do not invent owners or deadlines when none were assigned.
|
||||
- Do not silently discard older records that explain a changed commitment.
|
||||
- Do not send messages, create tasks, or update calendars unless the user separately authorizes those actions.
|
||||
|
||||
## Verification
|
||||
|
||||
- Every action has a source, status, and explicit or tentative owner and due date.
|
||||
- Conflicting values and revisions are resolved or visibly flagged.
|
||||
- The output covers decisions, actions, dependencies, risks, and open questions relevant to the request.
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: reviewable-external-draft
|
||||
description: "Reconcile source evidence and prepare an accurate external-facing draft without bypassing review"
|
||||
version: 1.0.0
|
||||
category: communication
|
||||
tags: [drafting, email, messages, review, reconciliation]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when preparing a client, customer, partner, leadership, or other external-facing update from internal messages or documents.
|
||||
|
||||
Do not use this procedure to send immediately unless the user explicitly authorizes the exact recipient and final content.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Confirm the audience, communication channel, requested tone, and whether the user asked for a draft or an immediate send.
|
||||
2. Gather the relevant source records and identify the latest values, dates, commitments, and unresolved discrepancies.
|
||||
3. Resolve recipient identity through the available contact source and avoid inferring internal versus external status from a display name alone.
|
||||
4. Draft only claims supported by the collected evidence; qualify uncertainty and omit internal-only detail that the audience should not receive.
|
||||
5. Save or present a reviewable draft through the native draft or document capability.
|
||||
6. Report the draft identifier or location plus any reconciliation notes that require human review.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not send a draft merely because a send-capable tool is available.
|
||||
- Do not copy stale figures when a later correction exists.
|
||||
- Do not conceal unresolved discrepancies behind polished prose.
|
||||
- Do not expose private internal discussion, credentials, or unrelated personal data.
|
||||
|
||||
## Verification
|
||||
|
||||
- Recipient identity and communication mode match the request.
|
||||
- Dates, figures, status, and commitments map to current source evidence.
|
||||
- The result remains reviewable unless an explicit send-now instruction authorized delivery.
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: scheduling-coordination
|
||||
description: "Coordinate availability, confirmations, calendar changes, and participant notifications with read-back verification"
|
||||
version: 1.0.0
|
||||
category: communication
|
||||
tags: [calendar, scheduling, coordination, availability]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when arranging or changing a meeting across multiple participants, calendars, time zones, or communication channels.
|
||||
|
||||
Do not create or modify an event when the user asked only for available options or a draft invitation.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Extract participants, duration, date range, time zones, location constraints, and required attendees.
|
||||
2. Resolve participant identities and inspect the relevant availability using declared calendar and contact capabilities.
|
||||
3. Compute candidate intervals in one explicit reference time zone and reject conflicts or insufficient travel buffers.
|
||||
4. Present or draft a small set of viable options when confirmation is still required.
|
||||
5. After authorization or recorded participant confirmation, create or update the event once with stable attendee identifiers.
|
||||
6. Read the event back and verify title, start, end, time zone, attendees, location, and conferencing details before drafting notifications.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not overwrite or cancel unrelated events to manufacture availability.
|
||||
- Do not mix local times without naming the time zone.
|
||||
- Do not treat a proposed time as confirmed.
|
||||
- Do not create duplicates when an existing event can be updated safely.
|
||||
|
||||
## Verification
|
||||
|
||||
- The selected interval satisfies duration, availability, and time-zone constraints.
|
||||
- The calendar read-back matches the authorized event details.
|
||||
- Notifications describe the same confirmed event and remain drafts unless sending was explicitly authorized.
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
name: support-triage-and-routing
|
||||
description: "Prioritize support requests, identify owners, route internally, and prepare safe customer drafts"
|
||||
version: 1.0.0
|
||||
category: communication
|
||||
tags: [support, triage, urgency, routing, drafts]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when reviewing a support backlog, identifying urgent incidents, assigning internal ownership, or drafting customer responses.
|
||||
|
||||
Do not use when the request is merely to summarize an unrelated inbox or when sender identity cannot be established safely.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Read each in-scope request in full and retain its stable message or ticket identifier.
|
||||
2. Resolve whether the sender is internal or external and identify the responsible internal team from available contacts and service ownership data.
|
||||
3. Classify urgency from impact and time sensitivity: critical for outage, data loss, security exposure, or imminent contractual breach; high for a blocked user without a workaround; medium for degraded service with a workaround; low for non-blocking inquiries.
|
||||
4. Record a concise problem statement, evidence, affected scope, workaround, owner, next action, and response deadline.
|
||||
5. Route internally only when the user has authorized operational messaging; prepare external responses as reviewable drafts by default.
|
||||
6. Re-read created assignments or drafts and produce an escalation summary grouped by urgency.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not infer severity from emotional language alone.
|
||||
- Do not expose one customer's data in another customer's response.
|
||||
- Do not send externally when the task calls for triage or drafting.
|
||||
- Do not mark an issue routed without a stable owner or observable routing result.
|
||||
|
||||
## Verification
|
||||
|
||||
- Every issue has a stable source identifier, urgency rationale, owner, and next action.
|
||||
- Critical and high items have explicit response targets and escalation state.
|
||||
- External communication is a draft unless the user explicitly authorized sending.
|
||||
@@ -0,0 +1,37 @@
|
||||
---
|
||||
name: developer-docs
|
||||
description: Find, read, and apply authoritative developer documentation during implementation
|
||||
version: 1.0.0
|
||||
category: dev
|
||||
tags: [docs, documentation, api, software-development]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-18T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when the user asks how a library, framework, API, protocol, CLI, or SDK works, or when implementation depends on version-specific behavior. Prefer this skill over guessing from memory.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Identify the exact product, package, version, and task. Ask one focused clarification only when the target is genuinely ambiguous.
|
||||
2. Prefer the vendor's or project's primary documentation, source repository, release notes, and API reference. Use a general search only to locate those sources.
|
||||
3. Read the relevant page or reference section, then apply the documented behavior to the user's codebase and active workspace.
|
||||
4. Separate documented facts from inference, and call out version or environment assumptions.
|
||||
5. For code changes, add a focused regression test for the documented contract and run it before reporting completion.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not present search snippets, stale cached knowledge, or a third-party tutorial as authoritative when primary documentation is available.
|
||||
- Do not silently mix instructions from different major versions.
|
||||
- Do not claim an API or option exists without confirming it in the relevant reference.
|
||||
- Do not use web search for a local project task when the active workspace and local tools can answer it.
|
||||
|
||||
## Verification
|
||||
|
||||
- The cited or retrieved documentation matches the target version.
|
||||
- The implementation or answer distinguishes source-backed facts from inference.
|
||||
- Any code change has a focused test or a concrete verification command.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: test-driven-development
|
||||
description: Build or fix software with a focused red-green-refactor loop
|
||||
version: 1.0.0
|
||||
category: general
|
||||
tags: [tdd, testing, debugging, red-green-refactor]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-18T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when implementing a feature, fixing a bug, or changing behavior where a regression test can define the expected result. Prefer this workflow for parser, routing, agent-loop, and UI behavior changes.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Inspect the relevant code, existing tests, and local conventions before editing.
|
||||
2. Write the smallest regression test that demonstrates the requested behavior or reproduces the bug.
|
||||
3. Run that test and confirm it fails for the expected reason, not because the test setup is broken.
|
||||
4. Make the smallest production change that makes the test pass.
|
||||
5. Run the focused test again, then run the surrounding module suite.
|
||||
6. Review the diff for unrelated changes, brittle assertions, hidden state, and missing error paths.
|
||||
7. Report the tests run and any remaining coverage or environment limits.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not write a test that only mirrors the implementation; assert the user-visible contract.
|
||||
- Do not weaken an assertion just to make a failing test pass.
|
||||
- Do not skip the focused failing-test step when the behavior is observable in a local test.
|
||||
- Keep network, filesystem, and model calls deterministic with fakes or fixtures unless the integration itself is under test.
|
||||
|
||||
## Verification
|
||||
|
||||
- The new regression test fails before the fix and passes after it.
|
||||
- The relevant focused suite passes.
|
||||
- The broader suite passes or its failure is explained with evidence.
|
||||
- The final diff contains the test and the production change needed for the same behavior.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: multimodal-evidence
|
||||
description: Extract and verify evidence from images, documents, and video without redundant inspection
|
||||
version: 1.0.1
|
||||
category: media
|
||||
tags: [image, video, document, evidence, ocr]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when the answer or requested artifact depends on visual, temporal, tabular, or textual evidence contained in images, documents, or video.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Identify the evidence required: objects, text, values, ordering, timestamps, labels, or visual relationships.
|
||||
2. Inspect the whole input or a broad representative sample first to establish structure and likely evidence locations.
|
||||
3. Narrow to relevant pages, frames, regions, or time intervals and record observations with their locations.
|
||||
4. Use the format's native parser for exact text and numbers: for example `python-docx` or ZIP/XML inspection for DOCX, `pdftotext` or a PDF library for PDF, spreadsheet readers for XLSX, and OCR only when the source is image-based. Do not search binary office files with plain `grep` or `cat`.
|
||||
5. Resolve conflicts with one targeted reinspection at better scale or a nearby frame rather than repeating the same crop.
|
||||
6. Build the answer or artifact from the evidence ledger and perform a final coverage check against every requested item.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not infer unseen content from filenames, surrounding text, or a single thumbnail.
|
||||
- Do not repeatedly inspect nearly identical regions without a new hypothesis.
|
||||
- Do not trust OCR blindly for small labels, punctuation, or numeric values.
|
||||
- Do not finalize before checking that every requested item has supporting evidence.
|
||||
|
||||
## Verification
|
||||
|
||||
- Each factual output can be traced to a page, frame, region, or timestamp.
|
||||
- Exact labels and numbers were visually checked after extraction.
|
||||
- The final response or artifact covers all requested evidence categories.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: web-research-fallback
|
||||
description: Research current web information with source-first search and controlled browser fallback
|
||||
version: 1.0.0
|
||||
category: research
|
||||
tags: [web, search, browser, sources, research]
|
||||
status: published
|
||||
confidence: 1.0
|
||||
source: builtin
|
||||
owner: ""
|
||||
created: "2026-08-30T00:00:00Z"
|
||||
---
|
||||
|
||||
## When to Use
|
||||
|
||||
Use when a task requires current public information, primary sources, multiple pages, or a site that cannot be reliably read from search results alone.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. Define the facts needed and the preferred primary source for each fact.
|
||||
2. Search with a focused query and use result metadata to select likely authoritative pages.
|
||||
3. Open the source directly and extract the relevant passage, date, and URL rather than relying on a search snippet.
|
||||
4. Use the private browser when the page requires interaction, client-side rendering, navigation, or visual inspection.
|
||||
5. If a page fails, try a primary-source alternative or a narrower route before broadening to secondary sources.
|
||||
6. Cross-check unstable or consequential claims and distinguish source-backed facts from inference.
|
||||
|
||||
## Pitfalls
|
||||
|
||||
- Do not treat snippets as evidence for claims not visible on the source page.
|
||||
- Do not browse repeatedly without recording what each page established.
|
||||
- Do not use a secondary summary when an accessible primary source answers the question.
|
||||
- Do not claim freshness without checking publication or update dates.
|
||||
|
||||
## Verification
|
||||
|
||||
- Each important claim maps to a source that directly supports it.
|
||||
- Time-sensitive facts include an observed date or version.
|
||||
- Browser interaction produced the needed page state or a documented fallback was used.
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Admin wipe route domain package (slice 2h, #4082/#4071).
|
||||
|
||||
Contains admin_wipe_routes.py, migrated from the flat routes/ directory.
|
||||
Backward-compat shim at routes/admin_wipe_routes.py re-exports from here.
|
||||
"""
|
||||
@@ -0,0 +1,176 @@
|
||||
"""Admin Danger Zone — per-category wipes.
|
||||
|
||||
Each endpoint is admin-only and truncates exactly one domain so the
|
||||
user can selectively reset memory / skills / notes / etc. without
|
||||
nuking everything. The catch-all `chats` endpoint mirrors the
|
||||
existing /api/sessions/all so the Danger Zone speaks one URL pattern.
|
||||
|
||||
URL shape: DELETE /api/admin/wipe/{kind}
|
||||
Kinds: chats, memory, skills, notes, tasks, documents, gallery, calendar.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
from fastapi import APIRouter, HTTPException, Request
|
||||
|
||||
from core.middleware import require_admin
|
||||
from core.database import (
|
||||
SessionLocal,
|
||||
Session as DbSession,
|
||||
ChatMessage as DbChatMessage,
|
||||
Memory,
|
||||
Note,
|
||||
ScheduledTask,
|
||||
TaskRun,
|
||||
Document,
|
||||
DocumentVersion,
|
||||
GalleryImage,
|
||||
GalleryAlbum,
|
||||
CalendarEvent,
|
||||
CalendarCal,
|
||||
)
|
||||
from src.constants import DATA_DIR, SKILLS_DIR, SKILLS_FILE, GALLERY_DIR, GALLERY_UPLOADS_DIR
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _wipe_memory_files():
|
||||
"""Blank memory.json + drop the per-owner tidy-state sidecar so the
|
||||
next audit doesn't try to diff against gone memories."""
|
||||
for name in ("memory.json", "memory_tidy_state.json"):
|
||||
p = os.path.join(DATA_DIR, name)
|
||||
if not os.path.exists(p):
|
||||
continue
|
||||
try:
|
||||
if name == "memory.json":
|
||||
with open(p, "w", encoding="utf-8") as f:
|
||||
json.dump([], f)
|
||||
else:
|
||||
os.remove(p)
|
||||
except OSError as e:
|
||||
logger.warning(f"Could not reset {name}: {e}")
|
||||
|
||||
|
||||
def _rmtree_quiet(path: str):
|
||||
"""rmtree that doesn't crash if the path doesn't exist."""
|
||||
if os.path.isdir(path):
|
||||
try:
|
||||
shutil.rmtree(path)
|
||||
except OSError as e:
|
||||
logger.warning(f"Could not remove {path}: {e}")
|
||||
|
||||
|
||||
def setup_admin_wipe_routes(session_manager):
|
||||
"""The session_manager is passed in so we can also clear its
|
||||
in-memory cache when wiping chats — without it the DB is empty
|
||||
but the next /api/sessions returns stale entries."""
|
||||
router = APIRouter(prefix="/api/admin")
|
||||
|
||||
@router.delete("/wipe/{kind}")
|
||||
def wipe(kind: str, request: Request):
|
||||
require_admin(request)
|
||||
kind = (kind or "").strip().lower()
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
if kind == "chats":
|
||||
count = db.query(DbSession).count()
|
||||
db.query(DbChatMessage).delete()
|
||||
db.query(DbSession).delete()
|
||||
db.commit()
|
||||
try:
|
||||
session_manager.sessions.clear()
|
||||
except Exception:
|
||||
pass
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "memory":
|
||||
count = db.query(Memory).count()
|
||||
db.query(Memory).delete()
|
||||
db.commit()
|
||||
_wipe_memory_files()
|
||||
# Drop the vector store too so semantic search doesn't
|
||||
# return ghosts. Lazy import — chromadb may not be
|
||||
# initialised in every deployment.
|
||||
try:
|
||||
from src.memory_vector import get_memory_vector_store
|
||||
mv = get_memory_vector_store()
|
||||
if mv and hasattr(mv, "clear"):
|
||||
mv.clear()
|
||||
except Exception as e:
|
||||
logger.info(f"Memory vector clear skipped: {e}")
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "skills":
|
||||
# Skills live as SKILL.md files under data/skills/. Drop
|
||||
# the entire directory; the SkillsManager re-creates the
|
||||
# tree on next write.
|
||||
skills_dir = SKILLS_DIR
|
||||
count = 0
|
||||
if os.path.isdir(skills_dir):
|
||||
# Count SKILL.md files for the response — quick walk.
|
||||
for _, _, files in os.walk(skills_dir):
|
||||
count += sum(1 for f in files if f == "SKILL.md")
|
||||
_rmtree_quiet(skills_dir)
|
||||
# Legacy fallback file
|
||||
legacy = SKILLS_FILE
|
||||
if os.path.exists(legacy):
|
||||
try:
|
||||
os.remove(legacy)
|
||||
except OSError:
|
||||
pass
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "notes":
|
||||
count = db.query(Note).count()
|
||||
db.query(Note).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "tasks":
|
||||
# TaskRun rows reference tasks via FK — clear them first.
|
||||
db.query(TaskRun).delete()
|
||||
count = db.query(ScheduledTask).count()
|
||||
db.query(ScheduledTask).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "documents":
|
||||
# DocumentVersion FKs Document — clear children first.
|
||||
db.query(DocumentVersion).delete()
|
||||
count = db.query(Document).count()
|
||||
db.query(Document).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "gallery":
|
||||
count = db.query(GalleryImage).count() + db.query(GalleryAlbum).count()
|
||||
db.query(GalleryImage).delete()
|
||||
db.query(GalleryAlbum).delete()
|
||||
db.commit()
|
||||
# Also drop the upload dir so disk doesn't keep orphans.
|
||||
_rmtree_quiet(GALLERY_DIR)
|
||||
_rmtree_quiet(GALLERY_UPLOADS_DIR)
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "calendar":
|
||||
# Events FK calendars — clear children first, then both.
|
||||
db.query(CalendarEvent).delete()
|
||||
count = db.query(CalendarCal).count()
|
||||
db.query(CalendarCal).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
raise HTTPException(400, f"Unknown wipe kind: {kind!r}")
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
db.rollback()
|
||||
logger.exception(f"Wipe {kind} failed")
|
||||
raise HTTPException(500, f"Wipe {kind} failed: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return router
|
||||
+12
-171
@@ -1,176 +1,17 @@
|
||||
"""Admin Danger Zone — per-category wipes.
|
||||
"""Backward-compat shim — canonical location is routes/admin_wipe/admin_wipe_routes.py.
|
||||
|
||||
Each endpoint is admin-only and truncates exactly one domain so the
|
||||
user can selectively reset memory / skills / notes / etc. without
|
||||
nuking everything. The catch-all `chats` endpoint mirrors the
|
||||
existing /api/sessions/all so the Danger Zone speaks one URL pattern.
|
||||
|
||||
URL shape: DELETE /api/admin/wipe/{kind}
|
||||
Kinds: chats, memory, skills, notes, tasks, documents, gallery, calendar.
|
||||
This module is replaced in ``sys.modules`` by the canonical module object so
|
||||
that ``import routes.admin_wipe_routes``, ``from routes.admin_wipe_routes
|
||||
import X``, ``importlib.import_module("routes.admin_wipe_routes")``, and the
|
||||
``import ... as admin_wipe_routes`` + ``monkeypatch.setattr(admin_wipe_routes,
|
||||
"SessionLocal", ...)`` / ``"require_admin"`` pattern used by
|
||||
test_admin_wipe_gallery.py all operate on the *same* object the application
|
||||
actually uses. Keeps existing import paths working after slice 2h
|
||||
(#4082/#4071).
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
from fastapi import APIRouter, HTTPException, Request
|
||||
import sys as _sys
|
||||
|
||||
from core.middleware import require_admin
|
||||
from core.database import (
|
||||
SessionLocal,
|
||||
Session as DbSession,
|
||||
ChatMessage as DbChatMessage,
|
||||
Memory,
|
||||
Note,
|
||||
ScheduledTask,
|
||||
TaskRun,
|
||||
Document,
|
||||
DocumentVersion,
|
||||
GalleryImage,
|
||||
GalleryAlbum,
|
||||
CalendarEvent,
|
||||
CalendarCal,
|
||||
)
|
||||
from src.constants import DATA_DIR, SKILLS_DIR, SKILLS_FILE, GALLERY_DIR, GALLERY_UPLOADS_DIR
|
||||
from routes.admin_wipe import admin_wipe_routes as _canonical # noqa: F401
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _wipe_memory_files():
|
||||
"""Blank memory.json + drop the per-owner tidy-state sidecar so the
|
||||
next audit doesn't try to diff against gone memories."""
|
||||
for name in ("memory.json", "memory_tidy_state.json"):
|
||||
p = os.path.join(DATA_DIR, name)
|
||||
if not os.path.exists(p):
|
||||
continue
|
||||
try:
|
||||
if name == "memory.json":
|
||||
with open(p, "w", encoding="utf-8") as f:
|
||||
json.dump([], f)
|
||||
else:
|
||||
os.remove(p)
|
||||
except OSError as e:
|
||||
logger.warning(f"Could not reset {name}: {e}")
|
||||
|
||||
|
||||
def _rmtree_quiet(path: str):
|
||||
"""rmtree that doesn't crash if the path doesn't exist."""
|
||||
if os.path.isdir(path):
|
||||
try:
|
||||
shutil.rmtree(path)
|
||||
except OSError as e:
|
||||
logger.warning(f"Could not remove {path}: {e}")
|
||||
|
||||
|
||||
def setup_admin_wipe_routes(session_manager):
|
||||
"""The session_manager is passed in so we can also clear its
|
||||
in-memory cache when wiping chats — without it the DB is empty
|
||||
but the next /api/sessions returns stale entries."""
|
||||
router = APIRouter(prefix="/api/admin")
|
||||
|
||||
@router.delete("/wipe/{kind}")
|
||||
def wipe(kind: str, request: Request):
|
||||
require_admin(request)
|
||||
kind = (kind or "").strip().lower()
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
if kind == "chats":
|
||||
count = db.query(DbSession).count()
|
||||
db.query(DbChatMessage).delete()
|
||||
db.query(DbSession).delete()
|
||||
db.commit()
|
||||
try:
|
||||
session_manager.sessions.clear()
|
||||
except Exception:
|
||||
pass
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "memory":
|
||||
count = db.query(Memory).count()
|
||||
db.query(Memory).delete()
|
||||
db.commit()
|
||||
_wipe_memory_files()
|
||||
# Drop the vector store too so semantic search doesn't
|
||||
# return ghosts. Lazy import — chromadb may not be
|
||||
# initialised in every deployment.
|
||||
try:
|
||||
from src.memory_vector import get_memory_vector_store
|
||||
mv = get_memory_vector_store()
|
||||
if mv and hasattr(mv, "clear"):
|
||||
mv.clear()
|
||||
except Exception as e:
|
||||
logger.info(f"Memory vector clear skipped: {e}")
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "skills":
|
||||
# Skills live as SKILL.md files under data/skills/. Drop
|
||||
# the entire directory; the SkillsManager re-creates the
|
||||
# tree on next write.
|
||||
skills_dir = SKILLS_DIR
|
||||
count = 0
|
||||
if os.path.isdir(skills_dir):
|
||||
# Count SKILL.md files for the response — quick walk.
|
||||
for _, _, files in os.walk(skills_dir):
|
||||
count += sum(1 for f in files if f == "SKILL.md")
|
||||
_rmtree_quiet(skills_dir)
|
||||
# Legacy fallback file
|
||||
legacy = SKILLS_FILE
|
||||
if os.path.exists(legacy):
|
||||
try:
|
||||
os.remove(legacy)
|
||||
except OSError:
|
||||
pass
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "notes":
|
||||
count = db.query(Note).count()
|
||||
db.query(Note).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "tasks":
|
||||
# TaskRun rows reference tasks via FK — clear them first.
|
||||
db.query(TaskRun).delete()
|
||||
count = db.query(ScheduledTask).count()
|
||||
db.query(ScheduledTask).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "documents":
|
||||
# DocumentVersion FKs Document — clear children first.
|
||||
db.query(DocumentVersion).delete()
|
||||
count = db.query(Document).count()
|
||||
db.query(Document).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "gallery":
|
||||
count = db.query(GalleryImage).count() + db.query(GalleryAlbum).count()
|
||||
db.query(GalleryImage).delete()
|
||||
db.query(GalleryAlbum).delete()
|
||||
db.commit()
|
||||
# Also drop the upload dir so disk doesn't keep orphans.
|
||||
_rmtree_quiet(GALLERY_DIR)
|
||||
_rmtree_quiet(GALLERY_UPLOADS_DIR)
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
if kind == "calendar":
|
||||
# Events FK calendars — clear children first, then both.
|
||||
db.query(CalendarEvent).delete()
|
||||
count = db.query(CalendarCal).count()
|
||||
db.query(CalendarCal).delete()
|
||||
db.commit()
|
||||
return {"status": "deleted", "kind": kind, "count": count}
|
||||
|
||||
raise HTTPException(400, f"Unknown wipe kind: {kind!r}")
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
db.rollback()
|
||||
logger.exception(f"Wipe {kind} failed")
|
||||
raise HTTPException(500, f"Wipe {kind} failed: {e}")
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return router
|
||||
_sys.modules[__name__] = _canonical
|
||||
|
||||
@@ -16,7 +16,7 @@ from pydantic import BaseModel
|
||||
|
||||
from core.database import SessionLocal, CrewMember, ScheduledTask
|
||||
from src.auth_helpers import get_current_user
|
||||
from core.auth import RESERVED_USERNAMES
|
||||
from src.owner_identity import REQUEST_SENTINEL_OWNERS
|
||||
from src.task_scheduler import compute_next_run
|
||||
|
||||
|
||||
@@ -90,11 +90,12 @@ def setup_assistant_routes(task_scheduler) -> APIRouter:
|
||||
# check-in tasks seeded. Hitting any /assistant route under one of these
|
||||
# used to seed a full CrewMember + Morning/Midday/Evening tasks under that
|
||||
# owner, which then double-fired alongside the real user's check-ins.
|
||||
# RESERVED_USERNAMES covers the same set; the `not owner` guard handles "".
|
||||
# REQUEST_SENTINEL_OWNERS covers request-only identities; Default/Local is a
|
||||
# reserved login name but remains a valid storage owner.
|
||||
|
||||
async def _get_or_create(owner: str) -> CrewMember:
|
||||
"""Return the per-owner assistant CrewMember, creating it on demand."""
|
||||
if not owner or owner in RESERVED_USERNAMES:
|
||||
if not owner or owner in REQUEST_SENTINEL_OWNERS:
|
||||
raise HTTPException(status_code=400, detail=f"Cannot seed assistant for {owner!r}")
|
||||
db = SessionLocal()
|
||||
try:
|
||||
|
||||
+88
-4
@@ -22,6 +22,8 @@ from src.settings import (
|
||||
load_features as _load_features,
|
||||
save_features as _save_features,
|
||||
DEFAULT_SETTINGS,
|
||||
RETIRED_SETTING_KEYS,
|
||||
without_retired_settings,
|
||||
)
|
||||
from src.integrations import (
|
||||
load_integrations,
|
||||
@@ -84,6 +86,33 @@ class SetOpenRegistrationRequest(BaseModel):
|
||||
SESSION_COOKIE = "odysseus_session"
|
||||
|
||||
|
||||
def _secure_cookie(request: Request) -> bool:
|
||||
"""Decide the ``Secure`` attribute of the session cookie.
|
||||
|
||||
``SECURE_COOKIES`` stays authoritative when it holds an explicit value:
|
||||
``true`` always marks the cookie Secure (the documented knob for a TLS
|
||||
proxy), ``false`` never does, which is the escape hatch for an install
|
||||
that still answers on plain HTTP alongside HTTPS. Anything else —
|
||||
unset, or the present-but-empty value docker-compose injects for a
|
||||
variable the host has not defined — derives it from the request, so an
|
||||
HTTPS login gets a Secure cookie without any configuration.
|
||||
|
||||
Either the connection scheme or ``X-Forwarded-Proto`` saying https is
|
||||
enough, which is the same test ``core/middleware.py`` applies before it
|
||||
sends HSTS. Uvicorn's proxy-headers middleware already folds that header
|
||||
into the scheme for the proxies it trusts, so reading it here only adds
|
||||
the case of a terminator that is not on a trusted address; the cost is
|
||||
that a client talking to the app directly can set the header and lock
|
||||
its own session out over plain HTTP.
|
||||
"""
|
||||
configured = os.getenv("SECURE_COOKIES", "").strip().lower()
|
||||
if configured in ("true", "false"):
|
||||
return configured == "true"
|
||||
# A chained proxy sends a list — the client-facing hop comes first.
|
||||
forwarded_proto = request.headers.get("x-forwarded-proto", "").split(",")[0]
|
||||
return request.url.scheme == "https" or forwarded_proto.strip().lower() == "https"
|
||||
|
||||
|
||||
def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
router = APIRouter(prefix="/api/auth", tags=["auth"])
|
||||
|
||||
@@ -157,7 +186,7 @@ def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
value=token,
|
||||
httponly=True,
|
||||
samesite="lax",
|
||||
secure=os.getenv("SECURE_COOKIES", "false").lower() == "true",
|
||||
secure=_secure_cookie(request),
|
||||
path="/",
|
||||
)
|
||||
if body.remember:
|
||||
@@ -345,9 +374,61 @@ def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
# docs, email accounts, tasks, etc.
|
||||
try:
|
||||
from sqlalchemy import func
|
||||
from core.database import Base, SessionLocal
|
||||
from core.database import (
|
||||
Base,
|
||||
EmailAccount,
|
||||
SessionLocal,
|
||||
lock_email_account_owner_mutations,
|
||||
)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
# Email-account defaults are protected by per-owner mutex rows.
|
||||
# A rename crosses two owner partitions, so lock both in the
|
||||
# shared helper's canonical order before inspecting either.
|
||||
lock_email_account_owner_mutations(
|
||||
db, old_username, new_username
|
||||
)
|
||||
|
||||
source_default_ids = [
|
||||
row[0]
|
||||
for row in (
|
||||
db.query(EmailAccount.id)
|
||||
.filter(
|
||||
func.lower(EmailAccount.owner) == old_username,
|
||||
EmailAccount.is_default == True, # noqa: E712
|
||||
)
|
||||
.order_by(EmailAccount.created_at.asc(), EmailAccount.id.asc())
|
||||
.all()
|
||||
)
|
||||
]
|
||||
destination_default_ids = [
|
||||
row[0]
|
||||
for row in (
|
||||
db.query(EmailAccount.id)
|
||||
.filter(
|
||||
func.lower(EmailAccount.owner) == new_username,
|
||||
EmailAccount.is_default == True, # noqa: E712
|
||||
)
|
||||
.order_by(EmailAccount.created_at.asc(), EmailAccount.id.asc())
|
||||
.all()
|
||||
)
|
||||
]
|
||||
if destination_default_ids:
|
||||
clear_default_ids = (
|
||||
destination_default_ids[1:] + source_default_ids
|
||||
)
|
||||
else:
|
||||
clear_default_ids = source_default_ids[1:]
|
||||
if clear_default_ids:
|
||||
(
|
||||
db.query(EmailAccount)
|
||||
.filter(EmailAccount.id.in_(clear_default_ids))
|
||||
.update(
|
||||
{EmailAccount.is_default: False},
|
||||
synchronize_session=False,
|
||||
)
|
||||
)
|
||||
|
||||
for mapper in Base.registry.mappers:
|
||||
model = mapper.class_
|
||||
if not hasattr(model, "owner"):
|
||||
@@ -637,7 +718,7 @@ def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
a scrubbed copy with secret keys blanked. The frontend uses this
|
||||
for keybinds + TTS prefs, so it stays callable without admin."""
|
||||
user = _get_current_user(request)
|
||||
settings = _load_settings()
|
||||
settings = without_retired_settings(_load_settings())
|
||||
if user and auth_manager.is_admin(user):
|
||||
return settings
|
||||
return scrub_settings(settings)
|
||||
@@ -655,8 +736,11 @@ def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
_INT_RANGES = {
|
||||
"agent_max_rounds": (1, 200),
|
||||
"agent_max_tool_calls": (0, 1000), # 0 = unlimited
|
||||
"auto_compact_threshold_percent": (50, 95),
|
||||
}
|
||||
for key in DEFAULT_SETTINGS:
|
||||
if key in RETIRED_SETTING_KEYS:
|
||||
continue
|
||||
if key not in body:
|
||||
continue
|
||||
val = body[key]
|
||||
@@ -669,7 +753,7 @@ def setup_auth_routes(auth_manager: AuthManager) -> APIRouter:
|
||||
val = max(lo, min(val, hi))
|
||||
current[key] = val
|
||||
_save_settings(current)
|
||||
return current
|
||||
return without_retired_settings(current)
|
||||
|
||||
# ---- Integrations CRUD ----
|
||||
|
||||
|
||||
+10
-1
@@ -6,6 +6,7 @@ from datetime import datetime
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Request, Response
|
||||
from core.middleware import require_admin
|
||||
from services.memory import MemoryStoreUnreadable
|
||||
from src.auth_helpers import get_current_user
|
||||
from src.settings import load_settings, save_settings, load_features, save_features
|
||||
|
||||
@@ -76,7 +77,15 @@ def setup_backup_routes(memory_manager, preset_manager, skills_manager) -> APIRo
|
||||
|
||||
# ── Memories ──
|
||||
if "memories" in body and isinstance(body["memories"], list):
|
||||
existing = memory_manager.load_all()
|
||||
# Strict load: importing on top of an unreadable store would write
|
||||
# only the incoming rows back and drop everything already saved.
|
||||
try:
|
||||
existing = memory_manager.load_all_for_update()
|
||||
except MemoryStoreUnreadable as e:
|
||||
logger.error("Refusing to import memories: %s", e)
|
||||
raise HTTPException(
|
||||
503, "Memory store is temporarily unreadable — nothing was imported."
|
||||
)
|
||||
# Dedup against THIS user's own memories only. Using every tenant's
|
||||
# rows (load_all) meant a memory whose text matched any other
|
||||
# user's was silently skipped, so the importing user lost their own
|
||||
|
||||
+373
-22
@@ -1,19 +1,22 @@
|
||||
"""Calendar routes — local SQLite-backed calendar CRUD."""
|
||||
|
||||
import logging
|
||||
import json
|
||||
import re
|
||||
import uuid
|
||||
from datetime import datetime, date, timedelta
|
||||
from datetime import datetime, date, timedelta, timezone
|
||||
from typing import Optional, List
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Request, UploadFile, File
|
||||
from pydantic import BaseModel
|
||||
from sqlalchemy import or_, and_
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
from dateutil.rrule import rrulestr
|
||||
|
||||
from core.database import SessionLocal, CalendarCal, CalendarDeletedEvent, CalendarEvent
|
||||
from src.auth_helpers import require_user
|
||||
from core.database import SessionLocal, CalendarCal, CalendarDeletedEvent, CalendarEvent, Note
|
||||
from src.auth_helpers import effective_user, require_user
|
||||
from src.upload_limits import read_upload_limited, ICS_MAX_BYTES
|
||||
from src.upload_handler import reserve_upload_references
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -204,6 +207,7 @@ class EventCreate(BaseModel):
|
||||
calendar_href: Optional[str] = None # calendar id
|
||||
rrule: Optional[str] = None
|
||||
color: Optional[str] = None # per-event color override
|
||||
reminder_minutes: Optional[int] = None
|
||||
|
||||
|
||||
class EventUpdate(BaseModel):
|
||||
@@ -215,26 +219,130 @@ class EventUpdate(BaseModel):
|
||||
location: Optional[str] = None
|
||||
rrule: Optional[str] = None
|
||||
color: Optional[str] = None
|
||||
reminder_minutes: Optional[int] = None
|
||||
|
||||
|
||||
# ── Helpers ──
|
||||
|
||||
_DEFAULT_CALENDAR_NAMESPACE = uuid.UUID("4840613a-9847-4a3b-bd75-19e6bc5fc3ce")
|
||||
|
||||
|
||||
def _default_calendar_id(owner: str, collision_index: int = 0) -> str:
|
||||
"""Return one stable primary-key candidate for an owner's lazy default.
|
||||
|
||||
Slot zero preserves the original owner-derived identifier. Later slots
|
||||
let a username be reused after its prior calendar was migrated to another
|
||||
owner during a rename, without making concurrent first use choose random
|
||||
and therefore divergent identifiers.
|
||||
"""
|
||||
if collision_index == 0:
|
||||
candidate_name = owner
|
||||
else:
|
||||
candidate_name = json.dumps(
|
||||
[owner, collision_index],
|
||||
ensure_ascii=False,
|
||||
separators=(",", ":"),
|
||||
)
|
||||
return str(uuid.uuid5(_DEFAULT_CALENDAR_NAMESPACE, candidate_name))
|
||||
|
||||
|
||||
def _begin_sqlite_default_write(db) -> None:
|
||||
"""Serialize an absent-default check with other SQLite writers.
|
||||
|
||||
SQLite's default deferred transactions allow two workers to both read an
|
||||
empty calendar set before either writes. ``BEGIN IMMEDIATE`` acquires the
|
||||
writer reservation before the second, authoritative lookup. We issue it
|
||||
only when the driver has not already opened a write transaction; a caller
|
||||
with a pending write already owns the required reservation.
|
||||
"""
|
||||
connection = db.connection()
|
||||
dbapi_connection = connection.connection
|
||||
driver_connection = getattr(
|
||||
dbapi_connection,
|
||||
"driver_connection",
|
||||
dbapi_connection,
|
||||
)
|
||||
if not getattr(driver_connection, "in_transaction", False):
|
||||
connection.exec_driver_sql("BEGIN IMMEDIATE")
|
||||
|
||||
|
||||
def _ensure_default_calendar(db, owner: str = None) -> CalendarCal:
|
||||
"""Create default calendar if none exist for this owner."""
|
||||
"""Return the owner's calendar, staging a default in the caller's transaction.
|
||||
|
||||
A stable owner-derived primary key makes concurrent first-use inserts
|
||||
converge on one row on every SQL backend. SQLite additionally serializes
|
||||
the absent-row check because its deferred transactions otherwise permit
|
||||
both workers to read the gap before either writes. Other backends recover
|
||||
a lost insert race inside a savepoint so the caller's event transaction
|
||||
remains usable and atomic.
|
||||
"""
|
||||
owner = owner or FALLBACK_OWNER
|
||||
cal = db.query(CalendarCal).filter(CalendarCal.owner == owner).first()
|
||||
if not cal:
|
||||
if cal:
|
||||
return cal
|
||||
|
||||
dialect = db.get_bind().dialect.name
|
||||
if dialect == "sqlite":
|
||||
_begin_sqlite_default_write(db)
|
||||
# Another worker may have committed while BEGIN IMMEDIATE waited.
|
||||
cal = db.query(CalendarCal).filter(CalendarCal.owner == owner).first()
|
||||
if cal:
|
||||
return cal
|
||||
|
||||
collision_index = 0
|
||||
while True:
|
||||
default_id = _default_calendar_id(owner, collision_index)
|
||||
|
||||
if dialect == "sqlite":
|
||||
# BEGIN IMMEDIATE above makes this occupancy check authoritative:
|
||||
# another SQLite writer cannot rename, delete, or claim this slot
|
||||
# until the caller commits or rolls back.
|
||||
occupant = db.query(CalendarCal).filter(
|
||||
CalendarCal.id == default_id,
|
||||
).first()
|
||||
if occupant is not None:
|
||||
if occupant.owner == owner:
|
||||
return occupant
|
||||
collision_index += 1
|
||||
continue
|
||||
|
||||
cal = CalendarCal(
|
||||
id=str(uuid.uuid4()),
|
||||
id=default_id,
|
||||
owner=owner,
|
||||
name="Personal",
|
||||
color="#5b8abf",
|
||||
source="local",
|
||||
)
|
||||
db.add(cal)
|
||||
db.commit()
|
||||
db.refresh(cal)
|
||||
return cal
|
||||
|
||||
if dialect == "sqlite":
|
||||
db.add(cal)
|
||||
db.flush()
|
||||
return cal
|
||||
|
||||
try:
|
||||
# A uniqueness failure rolls back only this savepoint, not an event
|
||||
# or reminder already staged by the caller's outer transaction.
|
||||
with db.begin_nested():
|
||||
db.add(cal)
|
||||
db.flush()
|
||||
return cal
|
||||
except IntegrityError:
|
||||
# Use a locking/current read so repeatable-read backends can observe
|
||||
# the row that won after our transaction's original empty snapshot.
|
||||
occupant = db.query(CalendarCal).filter(
|
||||
CalendarCal.id == default_id,
|
||||
).with_for_update().first()
|
||||
if occupant is None:
|
||||
# Do not misclassify an unrelated integrity failure as an ID
|
||||
# collision and loop forever. A concurrently deleted winner is
|
||||
# safe for the caller to retry as a fresh transaction.
|
||||
raise
|
||||
if occupant.owner == owner:
|
||||
return occupant
|
||||
# A renamed calendar owns this deterministic slot. Advance to the
|
||||
# next stable slot; concurrent callers for this owner will still
|
||||
# converge there.
|
||||
collision_index += 1
|
||||
|
||||
|
||||
# Per-request user time context. chat_routes sets this from browser timezone
|
||||
@@ -515,7 +623,133 @@ def _parse_dt(s: str) -> datetime:
|
||||
raise ValueError(f"could not parse datetime: {s!r}")
|
||||
|
||||
|
||||
def _event_to_dict(ev: CalendarEvent) -> dict:
|
||||
def _note_due_datetime(value: str | None) -> datetime | None:
|
||||
if not value:
|
||||
return None
|
||||
try:
|
||||
text = str(value).strip()
|
||||
if text.endswith("Z"):
|
||||
text = text[:-1] + "+00:00"
|
||||
due = datetime.fromisoformat(text)
|
||||
if due.tzinfo is not None:
|
||||
return due.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
return due
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _calendar_reminder_for_event(db, owner: str, ev: CalendarEvent) -> dict | None:
|
||||
"""Return the closest Notes reminder that belongs to this calendar event.
|
||||
|
||||
Calendar alarms are currently stored as Notes rows. Older rows do not carry
|
||||
an event UID, so match conservatively by the generated title plus due_date
|
||||
before the event start. This keeps existing reminder notes visible on the
|
||||
calendar without a schema migration.
|
||||
"""
|
||||
if not db or not owner or not ev or not ev.dtstart:
|
||||
return None
|
||||
summary = (ev.summary or "").strip()
|
||||
if not summary:
|
||||
return None
|
||||
|
||||
titles = [f"Calendar reminder: {summary}", f"Reminder: {summary}"]
|
||||
notes = (
|
||||
db.query(Note)
|
||||
.filter(
|
||||
Note.owner == owner,
|
||||
Note.archived == False, # noqa: E712
|
||||
Note.label == "calendar",
|
||||
Note.source == "calendar",
|
||||
Note.title.in_(titles),
|
||||
Note.due_date.isnot(None),
|
||||
)
|
||||
.all()
|
||||
)
|
||||
if not notes:
|
||||
return None
|
||||
|
||||
start = ev.dtstart
|
||||
if getattr(start, "tzinfo", None) is not None:
|
||||
start = start.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
best = None
|
||||
best_minutes = None
|
||||
for note in notes:
|
||||
due = _note_due_datetime(note.due_date)
|
||||
if due is None:
|
||||
continue
|
||||
minutes = round((start - due).total_seconds() / 60)
|
||||
if minutes < 0 or minutes > 7 * 24 * 60:
|
||||
continue
|
||||
if best is None or minutes < best_minutes:
|
||||
best = note
|
||||
best_minutes = minutes
|
||||
if best is None:
|
||||
return None
|
||||
return {
|
||||
"note_id": best.id,
|
||||
"due_date": best.due_date,
|
||||
"minutes": best_minutes,
|
||||
}
|
||||
|
||||
|
||||
def _delete_calendar_reminders_for_event(db, owner: str, ev: CalendarEvent) -> int:
|
||||
if not db or not owner or not ev:
|
||||
return 0
|
||||
summary = (ev.summary or "").strip()
|
||||
if not summary:
|
||||
return 0
|
||||
titles = [f"Calendar reminder: {summary}", f"Reminder: {summary}"]
|
||||
notes = (
|
||||
db.query(Note)
|
||||
.filter(
|
||||
Note.owner == owner,
|
||||
Note.archived == False, # noqa: E712
|
||||
Note.label == "calendar",
|
||||
Note.source == "calendar",
|
||||
Note.title.in_(titles),
|
||||
Note.due_date.isnot(None),
|
||||
)
|
||||
.all()
|
||||
)
|
||||
for note in notes:
|
||||
db.delete(note)
|
||||
return len(notes)
|
||||
|
||||
|
||||
def _create_calendar_reminder_for_event(db, owner: str, ev: CalendarEvent, minutes_before: int) -> dict:
|
||||
if not owner or not ev or not ev.dtstart:
|
||||
return {"note_id": None, "skipped_reason": "missing event"}
|
||||
minutes_before = max(0, int(minutes_before))
|
||||
start = ev.dtstart
|
||||
if getattr(start, "tzinfo", None) is not None:
|
||||
start = start.astimezone(timezone.utc).replace(tzinfo=None)
|
||||
remind_at = start - timedelta(minutes=minutes_before)
|
||||
now = datetime.utcnow() if getattr(ev, "is_utc", False) else datetime.now()
|
||||
if start <= now:
|
||||
return {"note_id": None, "skipped_reason": "event already passed"}
|
||||
if remind_at <= now:
|
||||
remind_at = now
|
||||
|
||||
summary = (ev.summary or "(no title)").strip() or "(no title)"
|
||||
location = (ev.location or "").strip()
|
||||
start_fmt = start.strftime("%a %b %d") if ev.all_day else start.strftime("%a %b %d %H:%M")
|
||||
loc = f" @ {location}" if location else ""
|
||||
due_date = remind_at.isoformat() + ("Z" if getattr(ev, "is_utc", False) and not ev.all_day else "")
|
||||
note = Note(
|
||||
id=str(uuid.uuid4()),
|
||||
owner=owner,
|
||||
title=f"Calendar reminder: {summary}",
|
||||
items=json.dumps([{"text": f"{summary}{loc} — {start_fmt}", "done": False, "checked": False}]),
|
||||
note_type="todo",
|
||||
label="calendar",
|
||||
due_date=due_date,
|
||||
source="calendar",
|
||||
)
|
||||
db.add(note)
|
||||
return {"note_id": note.id, "due_date": due_date, "minutes": minutes_before, "skipped_reason": None}
|
||||
|
||||
|
||||
def _event_to_dict(ev: CalendarEvent, db=None, owner: str | None = None) -> dict:
|
||||
"""Convert a CalendarEvent model to the API dict format.
|
||||
|
||||
Timed events whose stored datetimes represent UTC (is_utc=True) are
|
||||
@@ -531,6 +765,7 @@ def _event_to_dict(ev: CalendarEvent) -> dict:
|
||||
suffix = "Z" if getattr(ev, "is_utc", False) else ""
|
||||
start_str = ev.dtstart.isoformat() + suffix
|
||||
end_str = ev.dtend.isoformat() + suffix
|
||||
reminder = _calendar_reminder_for_event(db, owner, ev) if db and owner else None
|
||||
return {
|
||||
"uid": ev.uid,
|
||||
"summary": ev.summary or "",
|
||||
@@ -541,11 +776,16 @@ def _event_to_dict(ev: CalendarEvent) -> dict:
|
||||
"description": ev.description or "",
|
||||
"location": ev.location or "",
|
||||
"rrule": ev.rrule or "",
|
||||
"recurrence_exdates": _recurrence_exdates(ev),
|
||||
"calendar": ev.calendar.name if ev.calendar else "",
|
||||
"calendar_href": ev.calendar_id,
|
||||
"color": ev.color or (ev.calendar.color if ev.calendar else ""),
|
||||
"event_type": getattr(ev, "event_type", None),
|
||||
"importance": getattr(ev, "importance", None) or "normal",
|
||||
"has_reminder": bool(reminder),
|
||||
"reminder_note_id": reminder["note_id"] if reminder else None,
|
||||
"reminder_due_date": reminder["due_date"] if reminder else None,
|
||||
"reminder_minutes": reminder["minutes"] if reminder else None,
|
||||
}
|
||||
|
||||
|
||||
@@ -554,8 +794,30 @@ def _event_to_dict(ev: CalendarEvent) -> dict:
|
||||
_RRULE_EXPANSION_LIMIT = 1000
|
||||
|
||||
|
||||
def _recurrence_exdates(ev: CalendarEvent) -> list[str]:
|
||||
raw = getattr(ev, "recurrence_exdates", "") or ""
|
||||
if not raw:
|
||||
return []
|
||||
try:
|
||||
values = json.loads(raw)
|
||||
except Exception:
|
||||
return []
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
return [str(v) for v in values if isinstance(v, str) and v.strip()]
|
||||
|
||||
|
||||
def _occurrence_exdate_key(uid: str, ev: CalendarEvent) -> str:
|
||||
if "::" not in uid:
|
||||
return ""
|
||||
suffix = uid.split("::", 1)[1]
|
||||
if ev.all_day:
|
||||
return suffix[:10]
|
||||
return suffix[:16]
|
||||
|
||||
|
||||
def _expand_rrule(
|
||||
ev: CalendarEvent, start: datetime, end: datetime
|
||||
ev: CalendarEvent, start: datetime, end: datetime, db=None, owner: str | None = None
|
||||
) -> List[dict]:
|
||||
"""Expand a single recurring CalendarEvent into occurrence dicts.
|
||||
|
||||
@@ -573,7 +835,7 @@ def _expand_rrule(
|
||||
# Non-recurring — return the base event as-is. list_events
|
||||
# already filters non-recurring rows with the overlap check
|
||||
# in SQL, so we don't re-check here.
|
||||
d = _event_to_dict(ev)
|
||||
d = _event_to_dict(ev, db=db, owner=owner)
|
||||
d["is_recurrence"] = False
|
||||
d["series_uid"] = ev.uid
|
||||
d["truncated"] = False
|
||||
@@ -599,7 +861,7 @@ def _expand_rrule(
|
||||
logger.warning(
|
||||
"Failed to parse rrule=%r for event %s: %s", ev.rrule, ev.uid, ex
|
||||
)
|
||||
d = _event_to_dict(ev)
|
||||
d = _event_to_dict(ev, db=db, owner=owner)
|
||||
d["is_recurrence"] = False
|
||||
d["series_uid"] = ev.uid
|
||||
d["truncated"] = False
|
||||
@@ -617,7 +879,8 @@ def _expand_rrule(
|
||||
expand_start = start - duration
|
||||
results = []
|
||||
truncated = False
|
||||
base = _event_to_dict(ev)
|
||||
base = _event_to_dict(ev, db=db, owner=owner)
|
||||
exdates = set(_recurrence_exdates(ev))
|
||||
|
||||
for occ_start in rule.xafter(expand_start, inc=True):
|
||||
if occ_start >= end:
|
||||
@@ -638,8 +901,13 @@ def _expand_rrule(
|
||||
# Build the compound uid: {base_uid}::{date} or ::{datetime}
|
||||
if ev.all_day:
|
||||
occ_uid = f"{ev.uid}::{occ_start.strftime('%Y-%m-%d')}"
|
||||
exdate_key = occ_start.strftime("%Y-%m-%d")
|
||||
else:
|
||||
occ_uid = f"{ev.uid}::{occ_start.strftime('%Y-%m-%dT%H:%M')}"
|
||||
exdate_key = occ_start.strftime("%Y-%m-%dT%H:%M")
|
||||
|
||||
if exdate_key in exdates:
|
||||
continue
|
||||
|
||||
d = dict(base)
|
||||
d["uid"] = occ_uid
|
||||
@@ -667,9 +935,18 @@ def _expand_rrule(
|
||||
|
||||
# ── Routes ──
|
||||
|
||||
def setup_calendar_routes() -> APIRouter:
|
||||
def setup_calendar_routes(upload_handler=None) -> APIRouter:
|
||||
router = APIRouter(prefix="/api/calendar", tags=["calendar"])
|
||||
|
||||
def _reserve_calendar_uploads(request: Request, *values) -> None:
|
||||
missing_id = reserve_upload_references(
|
||||
upload_handler,
|
||||
effective_user(request),
|
||||
*values,
|
||||
)
|
||||
if missing_id:
|
||||
raise HTTPException(409, f"Referenced upload is no longer available: {missing_id}")
|
||||
|
||||
# ── CalDAV multi-account helpers ─────────────────────────────────────────
|
||||
|
||||
def _get_caldav_accounts(owner: str) -> list:
|
||||
@@ -883,7 +1160,24 @@ def setup_calendar_routes() -> APIRouter:
|
||||
'</d:prop></d:propfind>'
|
||||
)
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=8.0, follow_redirects=False, trust_env=False) as cx:
|
||||
# Build an SSL context that trusts the operator's custom CA bundle
|
||||
# (SSL_CERT_FILE / REQUESTS_CA_BUNDLE) so self-signed CalDAV servers
|
||||
# pass the pre-flight the same way they pass the real sync.
|
||||
# trust_env=False is kept to block proxy/auth env leakage; the CA
|
||||
# bundle is loaded explicitly instead.
|
||||
import ssl as _ssl
|
||||
_ssl_ctx = _ssl.create_default_context()
|
||||
# Disable VERIFY_X509_STRICT so certs without a keyUsage extension
|
||||
# (common in self-signed setups) are accepted, matching the
|
||||
# requests/urllib3 behavior used by the CalDAV sync path.
|
||||
_ssl_ctx.verify_flags &= ~_ssl.VERIFY_X509_STRICT
|
||||
_ca_bundle = _os.environ.get("SSL_CERT_FILE") or _os.environ.get("REQUESTS_CA_BUNDLE")
|
||||
if _ca_bundle:
|
||||
if _os.path.isfile(_ca_bundle):
|
||||
_ssl_ctx.load_verify_locations(_ca_bundle)
|
||||
else:
|
||||
logger.warning("CalDAV test: CA bundle %s not found, using system CAs", _ca_bundle)
|
||||
async with httpx.AsyncClient(timeout=8.0, follow_redirects=False, trust_env=False, verify=_ssl_ctx) as cx:
|
||||
r = await cx.request(
|
||||
"PROPFIND", url,
|
||||
auth=(user, pw),
|
||||
@@ -958,6 +1252,9 @@ def setup_calendar_routes() -> APIRouter:
|
||||
db = SessionLocal()
|
||||
try:
|
||||
_ensure_default_calendar(db, owner)
|
||||
# Listing calendars intentionally lazily creates a durable default.
|
||||
# Other callers commit it with the event they are creating.
|
||||
db.commit()
|
||||
cals = db.query(CalendarCal).filter(CalendarCal.owner == owner).all()
|
||||
return {"calendars": [
|
||||
{"name": c.name, "href": c.id, "color": c.color, "source": c.source}
|
||||
@@ -966,6 +1263,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
db.rollback()
|
||||
logger.error("Failed to list calendars: %s", e)
|
||||
raise HTTPException(500, "Failed to list calendars")
|
||||
finally:
|
||||
@@ -1020,7 +1318,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
# Expand recurring events into individual occurrences.
|
||||
expanded = []
|
||||
for e in events:
|
||||
expanded.extend(_expand_rrule(e, start_dt, end_dt))
|
||||
expanded.extend(_expand_rrule(e, start_dt, end_dt, db=db, owner=owner))
|
||||
|
||||
# Sort by occurrence start time for consistent frontend ordering.
|
||||
truncated = any(e.get("truncated") for e in expanded)
|
||||
@@ -1040,6 +1338,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
@router.post("/events")
|
||||
async def create_event(request: Request, data: EventCreate):
|
||||
owner = _require_user(request)
|
||||
_reserve_calendar_uploads(request, data.color, data.description, data.location)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
cal = None
|
||||
@@ -1085,10 +1384,19 @@ def setup_calendar_routes() -> APIRouter:
|
||||
caldav_sync_pending="create" if cal.source == "caldav" else None,
|
||||
)
|
||||
db.add(ev)
|
||||
reminder = None
|
||||
if data.reminder_minutes is not None:
|
||||
reminder = _create_calendar_reminder_for_event(db, owner, ev, data.reminder_minutes)
|
||||
db.commit()
|
||||
db.refresh(ev)
|
||||
if cal.source == "caldav":
|
||||
await _push_caldav_event_after_commit(owner, uid, "create")
|
||||
return {"ok": True, "uid": uid}
|
||||
return {
|
||||
"ok": True,
|
||||
"uid": uid,
|
||||
"event": _event_to_dict(ev, db=db, owner=owner),
|
||||
"reminder": reminder,
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1098,9 +1406,21 @@ def setup_calendar_routes() -> APIRouter:
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.get("/events/{uid}")
|
||||
async def get_event(request: Request, uid: str):
|
||||
owner = _require_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
base_uid = _resolve_base_uid(uid)
|
||||
ev = _get_or_404_event(db, base_uid, owner)
|
||||
return {"event": _event_to_dict(ev, db=db, owner=owner)}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.put("/events/{uid}")
|
||||
async def update_event(request: Request, uid: str, data: EventUpdate):
|
||||
owner = _require_user(request)
|
||||
_reserve_calendar_uploads(request, data.color, data.description, data.location)
|
||||
try:
|
||||
base_uid = _resolve_base_uid(uid)
|
||||
except ValueError as e:
|
||||
@@ -1133,13 +1453,24 @@ def setup_calendar_routes() -> APIRouter:
|
||||
ev.rrule = data.rrule
|
||||
if data.color is not None:
|
||||
ev.color = data.color if data.color else None
|
||||
reminder = None
|
||||
reminder_fields = getattr(data, "model_fields_set", getattr(data, "__fields_set__", set()))
|
||||
if "reminder_minutes" in reminder_fields:
|
||||
_delete_calendar_reminders_for_event(db, owner, ev)
|
||||
if data.reminder_minutes is not None:
|
||||
reminder = _create_calendar_reminder_for_event(db, owner, ev, data.reminder_minutes)
|
||||
is_caldav = ev.calendar and ev.calendar.source == "caldav"
|
||||
if is_caldav:
|
||||
ev.caldav_sync_pending = "update"
|
||||
db.commit()
|
||||
db.refresh(ev)
|
||||
if is_caldav:
|
||||
await _push_caldav_event_after_commit(owner, base_uid, "update")
|
||||
return {"ok": True}
|
||||
return {
|
||||
"ok": True,
|
||||
"event": _event_to_dict(ev, db=db, owner=owner),
|
||||
"reminder": reminder,
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1150,7 +1481,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
db.close()
|
||||
|
||||
@router.delete("/events/{uid}")
|
||||
async def delete_event(request: Request, uid: str):
|
||||
async def delete_event(request: Request, uid: str, scope: str = "series"):
|
||||
owner = _require_user(request)
|
||||
try:
|
||||
base_uid = _resolve_base_uid(uid)
|
||||
@@ -1159,9 +1490,27 @@ def setup_calendar_routes() -> APIRouter:
|
||||
db = SessionLocal()
|
||||
try:
|
||||
ev = _get_or_404_event(db, base_uid, owner)
|
||||
is_occurrence_delete = scope in {"occurrence", "instance"} and "::" in uid and bool(ev.rrule)
|
||||
is_caldav = ev.calendar and ev.calendar.source == "caldav"
|
||||
if scope in {"occurrence", "instance"} and not is_occurrence_delete:
|
||||
raise HTTPException(400, "Occurrence delete requires a recurring occurrence uid")
|
||||
if is_occurrence_delete:
|
||||
key = _occurrence_exdate_key(uid, ev)
|
||||
if not key:
|
||||
raise HTTPException(400, "Invalid recurring occurrence uid")
|
||||
exdates = _recurrence_exdates(ev)
|
||||
if key not in exdates:
|
||||
exdates.append(key)
|
||||
ev.recurrence_exdates = json.dumps(sorted(exdates))
|
||||
if is_caldav:
|
||||
ev.caldav_sync_pending = "update"
|
||||
db.commit()
|
||||
if is_caldav:
|
||||
await _push_caldav_event_after_commit(owner, base_uid, "update")
|
||||
return {"ok": True, "scope": "occurrence", "exdate": key}
|
||||
if is_caldav:
|
||||
_record_caldav_delete_tombstone(db, ev, owner)
|
||||
_delete_calendar_reminders_for_event(db, owner, ev)
|
||||
db.delete(ev)
|
||||
db.commit()
|
||||
if is_caldav:
|
||||
@@ -1179,6 +1528,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
@router.post("/calendars")
|
||||
async def create_calendar(request: Request, name: str = "Imported", color: str = "#5b8abf"):
|
||||
owner = _require_user(request)
|
||||
_reserve_calendar_uploads(request, color)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
cal = CalendarCal(
|
||||
@@ -1201,6 +1551,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
@router.put("/calendars/{cal_id}")
|
||||
async def update_calendar(request: Request, cal_id: str, name: str = None, color: str = None):
|
||||
owner = _require_user(request)
|
||||
_reserve_calendar_uploads(request, color)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
cal = _get_or_404_calendar(db, cal_id, owner)
|
||||
@@ -1239,7 +1590,7 @@ def setup_calendar_routes() -> APIRouter:
|
||||
raise HTTPException(400, f"Invalid ICS file: {e}")
|
||||
|
||||
# Sanitize display name — length cap + strip control chars
|
||||
raw_name = calendar_name.strip() or (file.filename or "").replace(".ics", "").replace("_", " ").strip() or "Imported"
|
||||
raw_name = calendar_name.strip() or re.sub(r"\.(?:calendar|ics|ical)$", "", file.filename or "", flags=re.IGNORECASE).replace("_", " ").strip() or "Imported"
|
||||
cal_display = "".join(c for c in raw_name if c.isprintable())[:120] or "Imported"
|
||||
|
||||
target_cal = db.query(CalendarCal).filter(
|
||||
|
||||
+620
-140
File diff suppressed because it is too large
Load Diff
+3089
-146
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,5 @@
|
||||
"""Cleanup route domain package (slice 2g, #4082/#4071).
|
||||
|
||||
Contains cleanup_routes.py, migrated from the flat routes/ directory.
|
||||
Backward-compat shim at routes/cleanup_routes.py re-exports from here.
|
||||
"""
|
||||
@@ -0,0 +1,60 @@
|
||||
# routes/cleanup_routes.py
|
||||
"""Routes for cleanup operations."""
|
||||
import logging
|
||||
from fastapi import APIRouter, HTTPException, Request
|
||||
from src.cleanup_service import get_cleanup_preview, cleanup_sessions
|
||||
from src.auth_helpers import get_current_user
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
def setup_cleanup_routes(session_manager):
|
||||
"""
|
||||
Setup cleanup-related routes.
|
||||
|
||||
Args:
|
||||
session_manager: SessionManager instance
|
||||
|
||||
Returns:
|
||||
APIRouter instance with cleanup routes
|
||||
"""
|
||||
router = APIRouter(prefix="/api/cleanup")
|
||||
|
||||
@router.get("/preview")
|
||||
async def cleanup_preview(request: Request):
|
||||
"""
|
||||
Preview what would be cleaned up without making any changes.
|
||||
|
||||
Returns:
|
||||
JSON response with lists of sessions that would be archived/deleted and estimated space savings
|
||||
"""
|
||||
user = get_current_user(request)
|
||||
try:
|
||||
preview = await get_cleanup_preview(owner=user)
|
||||
return preview
|
||||
except Exception as e:
|
||||
logger.error(f"Cleanup preview failed: {e}")
|
||||
raise HTTPException(500, "Cleanup preview generation failed")
|
||||
|
||||
@router.post("")
|
||||
async def cleanup_endpoint(request: Request):
|
||||
"""
|
||||
Perform cleanup operations:
|
||||
1. Archive inactive sessions (not accessed for 7 days)
|
||||
2. Delete old sessions (archived, not important, not accessed for 14+ days, with fewer than 10 messages)
|
||||
|
||||
Returns:
|
||||
JSON response with counts of deleted and archived sessions, and space freed
|
||||
"""
|
||||
user = get_current_user(request)
|
||||
try:
|
||||
archived_count, deleted_count, space_freed_mb = await cleanup_sessions(session_manager, owner=user)
|
||||
return {
|
||||
"archived_count": archived_count,
|
||||
"deleted_count": deleted_count,
|
||||
"space_freed_mb": round(space_freed_mb, 2)
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"Cleanup failed: {e}")
|
||||
raise HTTPException(500, "Cleanup operation failed")
|
||||
|
||||
return router
|
||||
+13
-56
@@ -1,60 +1,17 @@
|
||||
# routes/cleanup_routes.py
|
||||
"""Routes for cleanup operations."""
|
||||
import logging
|
||||
from fastapi import APIRouter, HTTPException, Request
|
||||
from src.cleanup_service import get_cleanup_preview, cleanup_sessions
|
||||
from src.auth_helpers import get_current_user
|
||||
"""Backward-compat shim — canonical location is routes/cleanup/cleanup_routes.py.
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
This module is replaced in ``sys.modules`` by the canonical module object so
|
||||
that ``import routes.cleanup_routes``, ``from routes.cleanup_routes import X``,
|
||||
``importlib.import_module("routes.cleanup_routes")``, and the string-targeted
|
||||
``monkeypatch.setattr("routes.cleanup_routes.get_cleanup_preview", ...)`` /
|
||||
``"routes.cleanup_routes.get_current_user"`` / ``"routes.cleanup_routes.
|
||||
cleanup_sessions"`` pattern used by test_cleanup_owner_scope.py all operate
|
||||
on the *same* object the application actually uses. Keeps existing import
|
||||
paths working after slice 2g (#4082/#4071).
|
||||
"""
|
||||
|
||||
def setup_cleanup_routes(session_manager):
|
||||
"""
|
||||
Setup cleanup-related routes.
|
||||
import sys as _sys
|
||||
|
||||
Args:
|
||||
session_manager: SessionManager instance
|
||||
from routes.cleanup import cleanup_routes as _canonical # noqa: F401
|
||||
|
||||
Returns:
|
||||
APIRouter instance with cleanup routes
|
||||
"""
|
||||
router = APIRouter(prefix="/api/cleanup")
|
||||
|
||||
@router.get("/preview")
|
||||
async def cleanup_preview(request: Request):
|
||||
"""
|
||||
Preview what would be cleaned up without making any changes.
|
||||
|
||||
Returns:
|
||||
JSON response with lists of sessions that would be archived/deleted and estimated space savings
|
||||
"""
|
||||
user = get_current_user(request)
|
||||
try:
|
||||
preview = await get_cleanup_preview(owner=user)
|
||||
return preview
|
||||
except Exception as e:
|
||||
logger.error(f"Cleanup preview failed: {e}")
|
||||
raise HTTPException(500, "Cleanup preview generation failed")
|
||||
|
||||
@router.post("")
|
||||
async def cleanup_endpoint(request: Request):
|
||||
"""
|
||||
Perform cleanup operations:
|
||||
1. Archive inactive sessions (not accessed for 7 days)
|
||||
2. Delete old sessions (archived, not important, not accessed for 14+ days, with fewer than 10 messages)
|
||||
|
||||
Returns:
|
||||
JSON response with counts of deleted and archived sessions, and space freed
|
||||
"""
|
||||
user = get_current_user(request)
|
||||
try:
|
||||
archived_count, deleted_count, space_freed_mb = await cleanup_sessions(session_manager, owner=user)
|
||||
return {
|
||||
"archived_count": archived_count,
|
||||
"deleted_count": deleted_count,
|
||||
"space_freed_mb": round(space_freed_mb, 2)
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"Cleanup failed: {e}")
|
||||
raise HTTPException(500, "Cleanup operation failed")
|
||||
|
||||
return router
|
||||
_sys.modules[__name__] = _canonical
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Compare route domain package (slice 2i, #4082/#4071).
|
||||
|
||||
Contains compare_routes.py, migrated from the flat routes/ directory.
|
||||
Backward-compat shim at routes/compare_routes.py re-exports from here.
|
||||
"""
|
||||
@@ -0,0 +1,365 @@
|
||||
# routes/compare_routes.py
|
||||
"""Model A/B comparison routes."""
|
||||
import json
|
||||
import uuid
|
||||
import random
|
||||
from datetime import datetime
|
||||
from fastapi import APIRouter, Form, HTTPException, Request
|
||||
from typing import List
|
||||
from pydantic import BaseModel
|
||||
import logging
|
||||
|
||||
from core.database import Comparison, SessionLocal
|
||||
from core.session_manager import SessionManager
|
||||
from src.auth_helpers import get_current_user
|
||||
from routes.session_routes import _reject_raw_endpoint_url_for_non_admin
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/api/compare", tags=["compare"])
|
||||
|
||||
|
||||
def _owned_endpoint_by_url(db, base_url, owner):
|
||||
"""ModelEndpoint whose base_url == `base_url` and is VISIBLE to `owner`
|
||||
(their own rows + legacy null-owner "shared" rows); None otherwise.
|
||||
|
||||
Owner-scoped on purpose. ModelEndpoint is per-user (core/database.py: non-null
|
||||
owner = private, "the model picker only shows the endpoint to that user") and
|
||||
holds a decrypted `api_key`. start_comparison copies the matched row's api_key
|
||||
into the caller-owned [CMP] session's headers, which then drives that session's
|
||||
/api/chat_stream calls — so an UNSCOPED base_url match would let a user mint a
|
||||
comparison bound to ANOTHER user's private endpoint and spend that owner's
|
||||
api_key / reach whatever base_url they configured. Mirrors
|
||||
session_routes._owned_endpoint. A null/empty owner is a no-op (single-user /
|
||||
legacy mode).
|
||||
"""
|
||||
from core.database import ModelEndpoint
|
||||
from src.auth_helpers import owner_filter
|
||||
q = db.query(ModelEndpoint).filter(ModelEndpoint.base_url == base_url)
|
||||
return owner_filter(q, ModelEndpoint, owner).first()
|
||||
|
||||
|
||||
def _owned_endpoint_by_id(db, endpoint_id, owner):
|
||||
"""ModelEndpoint whose id == `endpoint_id` and is VISIBLE to `owner` (their
|
||||
own rows + legacy null-owner "shared" rows); None otherwise.
|
||||
|
||||
Preferred over _owned_endpoint_by_url for credential resolution: two visible
|
||||
endpoints can share the same base_url but hold DIFFERENT api_keys (e.g. two
|
||||
accounts on the same provider). A base_url-only match returns whichever row
|
||||
sorts first, so it can copy the WRONG owner-scoped key into the [CMP] session.
|
||||
An id pins the exact registered endpoint, so /api/compare/start prefers it and
|
||||
only falls back to URL matching for legacy / admin raw-URL callers. Owner
|
||||
scoping is identical to _owned_endpoint_by_url (a null/empty owner is a no-op).
|
||||
"""
|
||||
from core.database import ModelEndpoint
|
||||
from src.auth_helpers import owner_filter
|
||||
q = db.query(ModelEndpoint).filter(ModelEndpoint.id == endpoint_id)
|
||||
return owner_filter(q, ModelEndpoint, owner).first()
|
||||
|
||||
|
||||
class RecordVoteRequest(BaseModel):
|
||||
prompt: str
|
||||
models: List[str]
|
||||
winner: str # model name or "tie"
|
||||
is_blind: bool = True
|
||||
|
||||
|
||||
def setup_compare_routes(session_manager: SessionManager):
|
||||
"""Setup comparison routes."""
|
||||
|
||||
@router.post("/start")
|
||||
def start_comparison(
|
||||
request: Request,
|
||||
prompt: str = Form(...),
|
||||
model_a: str = Form(...),
|
||||
model_b: str = Form(...),
|
||||
endpoint_a: str = Form(""),
|
||||
endpoint_b: str = Form(""),
|
||||
endpoint_a_id: str = Form(""),
|
||||
endpoint_b_id: str = Form(""),
|
||||
is_blind: str = Form("true"),
|
||||
):
|
||||
"""Create two ephemeral sessions and a comparison record.
|
||||
|
||||
Returns the comparison ID and the two session IDs so the client
|
||||
can fire two independent SSE streams to /api/chat_stream.
|
||||
"""
|
||||
user = getattr(request.state, 'current_user', None)
|
||||
comp_id = str(uuid.uuid4())
|
||||
sid_a = str(uuid.uuid4())
|
||||
sid_b = str(uuid.uuid4())
|
||||
|
||||
# Blind mapping: randomly assign left/right
|
||||
blind = str(is_blind).lower() == "true"
|
||||
if blind:
|
||||
mapping = {"left": "a", "right": "b"}
|
||||
if random.random() > 0.5:
|
||||
mapping = {"left": "b", "right": "a"}
|
||||
else:
|
||||
mapping = {"left": "a", "right": "b"}
|
||||
|
||||
# Map session IDs to left/right based on blind mapping
|
||||
session_left = sid_a if mapping["left"] == "a" else sid_b
|
||||
session_right = sid_a if mapping["right"] == "a" else sid_b
|
||||
|
||||
# In blind mode, name the helper sessions by their neutral slot
|
||||
# ("Model A" / "Model B") instead of the real model. Otherwise the
|
||||
# session name leaks the model in the sidebar and GET /api/sessions,
|
||||
# de-anonymizing the comparison before the user votes (issue #1285).
|
||||
slot_name = {session_left: "Model A", session_right: "Model B"}
|
||||
|
||||
# SECURITY: resolve and validate BOTH endpoints before creating any
|
||||
# session. Compare copies a registered endpoint's Authorization header
|
||||
# into the [CMP] session, so validating one endpoint while creating its
|
||||
# session, then rejecting the other, would leave a partial compare
|
||||
# session behind with that header attached. Doing all the owner-scope
|
||||
# resolution + raw-URL rejection up front means a 403 on either endpoint
|
||||
# aborts the whole request with nothing created and no header copied.
|
||||
from src.endpoint_resolver import build_chat_url, build_headers, normalize_base
|
||||
resolved = []
|
||||
db = SessionLocal()
|
||||
try:
|
||||
for sid, model, endpoint, endpoint_id in [
|
||||
(sid_a, model_a, endpoint_a, endpoint_a_id),
|
||||
(sid_b, model_b, endpoint_b, endpoint_b_id),
|
||||
]:
|
||||
# Prefer an explicit endpoint id: it pins the EXACT registered
|
||||
# endpoint (and its api_key), even when two endpoints visible to
|
||||
# the caller share a base_url with different keys — a URL-only
|
||||
# match would copy whichever row sorts first, i.e. possibly the
|
||||
# wrong key. Fall back to URL resolution only for legacy / admin
|
||||
# raw-URL callers that don't send an id.
|
||||
eid = endpoint_id.strip() if isinstance(endpoint_id, str) else ""
|
||||
if eid:
|
||||
ep = _owned_endpoint_by_id(db, eid, user)
|
||||
if ep is None:
|
||||
# An id the caller can't see (wrong owner / deleted) must
|
||||
# NOT silently fall back to a same-URL row with a different
|
||||
# key — that's exactly the mix-up ids exist to prevent.
|
||||
raise HTTPException(404, "Model endpoint not found")
|
||||
# The id already resolved the endpoint; ignore any raw URL the
|
||||
# caller also sent and dial the stored config instead.
|
||||
endpoint = ep.base_url
|
||||
elif not endpoint:
|
||||
raise HTTPException(
|
||||
422, "endpoint_a/endpoint_b or endpoint_a_id/endpoint_b_id is required"
|
||||
)
|
||||
else:
|
||||
# Resolve the supplied URL to a ModelEndpoint the caller owns
|
||||
# (their own rows + legacy null-owner shared rows), scoped so a
|
||||
# comparison can't borrow another user's private endpoint key.
|
||||
base = normalize_base(endpoint)
|
||||
ep = _owned_endpoint_by_url(db, base, user)
|
||||
# Reject *unregistered* raw URLs for signed-in non-admins; a
|
||||
# matched registered endpoint supplies an id so the caller can
|
||||
# still compare endpoints they own. Blanket-rejecting here (the
|
||||
# earlier `endpoint_id=None` call) locked non-admins out of
|
||||
# compare entirely, since compare resolves endpoints by URL with
|
||||
# no endpoint_id. Mirrors the gallery inpaint/harmonize checks.
|
||||
# Raised here (phase 1), before any session exists.
|
||||
_reject_raw_endpoint_url_for_non_admin(
|
||||
request, user, str(ep.id) if ep is not None else None, endpoint
|
||||
)
|
||||
# Bind the [CMP] session to the RESOLVED endpoint, not the raw
|
||||
# caller-supplied string. When the URL matches a registered
|
||||
# endpoint visible to the caller, use that row's own normalized
|
||||
# base URL (the same value owner scoping + endpoint validation
|
||||
# already vetted) so the session dials exactly where the stored
|
||||
# config points. The raw `endpoint` only survives for callers
|
||||
# allowed to pass one — admins / single-user mode, where
|
||||
# `_reject_raw_endpoint_url_for_non_admin` is a no-op and `ep`
|
||||
# is None. Mirrors the registered-endpoint path in session_routes.
|
||||
session_endpoint_url = (
|
||||
build_chat_url(normalize_base(ep.base_url)) if ep is not None else endpoint
|
||||
)
|
||||
# Headers come only from a matched endpoint's key; None when
|
||||
# `ep` is None (raw admin URL or no match), so a comparison can
|
||||
# never inherit another user's key/headers.
|
||||
headers = build_headers(ep.api_key, ep.base_url) if (ep and ep.api_key) else None
|
||||
resolved.append((sid, model, session_endpoint_url, headers))
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
# Both endpoints validated — only now create the ephemeral [CMP]
|
||||
# sessions and copy any resolved headers.
|
||||
for sid, model, session_endpoint_url, headers in resolved:
|
||||
name = f"[CMP] {slot_name[sid]}" if blind else f"[CMP] {model.split('/')[-1]}"
|
||||
session_manager.create_session(
|
||||
session_id=sid,
|
||||
name=name,
|
||||
endpoint_url=session_endpoint_url,
|
||||
model=model,
|
||||
rag=False,
|
||||
owner=user,
|
||||
)
|
||||
if headers:
|
||||
s = session_manager.sessions.get(sid)
|
||||
if s:
|
||||
s.headers = headers
|
||||
|
||||
# Store comparison record
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = Comparison(
|
||||
id=comp_id,
|
||||
prompt=prompt,
|
||||
model_a=model_a,
|
||||
model_b=model_b,
|
||||
# Record the URL the session actually dials. For URL callers this
|
||||
# is their raw input; for id-only callers (empty endpoint_a/_b)
|
||||
# fall back to the resolved endpoint URL so the column stays
|
||||
# meaningful and non-null. resolved is in [a, b] order.
|
||||
endpoint_a=endpoint_a or resolved[0][2],
|
||||
endpoint_b=endpoint_b or resolved[1][2],
|
||||
is_blind=blind,
|
||||
blind_mapping=json.dumps(mapping),
|
||||
owner=user,
|
||||
)
|
||||
db.add(comp)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
# In blind mode, withhold the model identities AND the left/right
|
||||
# mapping from the response. The client already knows model_a/model_b
|
||||
# (it sent them), so returning either would defeat blind mode. They are
|
||||
# revealed by POST /api/compare/{id}/vote once the user has voted (#1285).
|
||||
return {
|
||||
"id": comp_id,
|
||||
"session_left": session_left,
|
||||
"session_right": session_right,
|
||||
"model_left": None if blind else (model_a if mapping["left"] == "a" else model_b),
|
||||
"model_right": None if blind else (model_a if mapping["right"] == "a" else model_b),
|
||||
"is_blind": blind,
|
||||
"mapping": None if blind else mapping,
|
||||
}
|
||||
|
||||
@router.post("/{comp_id}/vote")
|
||||
def vote_comparison(
|
||||
request: Request,
|
||||
comp_id: str,
|
||||
winner: str = Form(...), # "left", "right", or "tie"
|
||||
):
|
||||
"""Record the user's vote and reveal model names if blind."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = db.query(Comparison).filter(Comparison.id == comp_id).first()
|
||||
if not comp:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
# SECURITY: strict ownership — null-owner Comparisons were
|
||||
# accessible to every user.
|
||||
if user and comp.owner != user:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
if comp.winner:
|
||||
raise HTTPException(400, "Already voted")
|
||||
|
||||
mapping = json.loads(comp.blind_mapping) if comp.blind_mapping else {"left": "a", "right": "b"}
|
||||
|
||||
if winner == "tie":
|
||||
comp.winner = "tie"
|
||||
elif winner == "left":
|
||||
comp.winner = mapping["left"]
|
||||
elif winner == "right":
|
||||
comp.winner = mapping["right"]
|
||||
else:
|
||||
raise HTTPException(400, "winner must be 'left', 'right', or 'tie'")
|
||||
|
||||
comp.voted_at = datetime.utcnow()
|
||||
db.commit()
|
||||
|
||||
return {
|
||||
"winner": comp.winner,
|
||||
"model_a": comp.model_a,
|
||||
"model_b": comp.model_b,
|
||||
"revealed": {
|
||||
"left": comp.model_a if mapping["left"] == "a" else comp.model_b,
|
||||
"right": comp.model_a if mapping["right"] == "a" else comp.model_b,
|
||||
},
|
||||
}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.post("/record")
|
||||
def record_comparison(request: Request, body: RecordVoteRequest):
|
||||
"""Lightweight endpoint to record a comparison vote from the frontend."""
|
||||
user = get_current_user(request)
|
||||
comp_id = str(uuid.uuid4())
|
||||
|
||||
model_a = body.models[0] if len(body.models) > 0 else ""
|
||||
model_b = body.models[1] if len(body.models) > 1 else ""
|
||||
|
||||
# For N>2 models, store the full list as JSON in blind_mapping
|
||||
if len(body.models) > 2:
|
||||
blind_mapping = json.dumps({"models": body.models})
|
||||
else:
|
||||
blind_mapping = None
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = Comparison(
|
||||
id=comp_id,
|
||||
prompt=body.prompt[:500],
|
||||
model_a=model_a,
|
||||
model_b=model_b,
|
||||
endpoint_a="",
|
||||
endpoint_b="",
|
||||
winner=body.winner,
|
||||
is_blind=body.is_blind,
|
||||
blind_mapping=blind_mapping,
|
||||
voted_at=datetime.utcnow(),
|
||||
owner=user,
|
||||
)
|
||||
db.add(comp)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return {"status": "ok", "id": comp_id}
|
||||
|
||||
@router.get("/history")
|
||||
def list_comparisons(request: Request):
|
||||
"""List past comparisons."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
q = db.query(Comparison)
|
||||
if user:
|
||||
q = q.filter(Comparison.owner == user)
|
||||
comps = q.order_by(Comparison.created_at.desc()).limit(50).all()
|
||||
return [
|
||||
{
|
||||
"id": c.id,
|
||||
"prompt": c.prompt[:100],
|
||||
"model_a": c.model_a,
|
||||
"model_b": c.model_b,
|
||||
"winner": c.winner,
|
||||
"is_blind": c.is_blind,
|
||||
"voted_at": c.voted_at.isoformat() if c.voted_at else None,
|
||||
"created_at": c.created_at.isoformat() if c.created_at else None,
|
||||
}
|
||||
for c in comps
|
||||
]
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.delete("/{comp_id}")
|
||||
def delete_comparison(request: Request, comp_id: str):
|
||||
"""Delete a comparison and its ephemeral sessions."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = db.query(Comparison).filter(Comparison.id == comp_id).first()
|
||||
if not comp:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
# SECURITY: strict ownership — null-owner Comparisons were
|
||||
# accessible to every user.
|
||||
if user and comp.owner != user:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
db.delete(comp)
|
||||
db.commit()
|
||||
return {"status": "deleted"}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return router
|
||||
+14
-361
@@ -1,365 +1,18 @@
|
||||
# routes/compare_routes.py
|
||||
"""Model A/B comparison routes."""
|
||||
import json
|
||||
import uuid
|
||||
import random
|
||||
from datetime import datetime
|
||||
from fastapi import APIRouter, Form, HTTPException, Request
|
||||
from typing import List
|
||||
from pydantic import BaseModel
|
||||
import logging
|
||||
"""Backward-compat shim — canonical location is routes/compare/compare_routes.py.
|
||||
|
||||
from core.database import Comparison, SessionLocal
|
||||
from core.session_manager import SessionManager
|
||||
from src.auth_helpers import get_current_user
|
||||
from routes.session_routes import _reject_raw_endpoint_url_for_non_admin
|
||||
This module is replaced in ``sys.modules`` by the canonical module object so
|
||||
that ``import routes.compare_routes``, ``from routes.compare_routes import X``,
|
||||
``importlib.import_module("routes.compare_routes")``, and the
|
||||
``import ... as cr`` + ``monkeypatch.setattr(cr, "SessionLocal", ...)`` /
|
||||
``"_owned_endpoint_by_url"`` / ``"_owned_endpoint_by_id"`` pattern used by
|
||||
test_endpoint_owner_scope_followup.py all operate on the *same* object the
|
||||
application actually uses. Keeps existing import paths working after
|
||||
slice 2i (#4082/#4071). Source-introspection tests read the canonical file
|
||||
by path.
|
||||
"""
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
import sys as _sys
|
||||
|
||||
router = APIRouter(prefix="/api/compare", tags=["compare"])
|
||||
from routes.compare import compare_routes as _canonical # noqa: F401
|
||||
|
||||
|
||||
def _owned_endpoint_by_url(db, base_url, owner):
|
||||
"""ModelEndpoint whose base_url == `base_url` and is VISIBLE to `owner`
|
||||
(their own rows + legacy null-owner "shared" rows); None otherwise.
|
||||
|
||||
Owner-scoped on purpose. ModelEndpoint is per-user (core/database.py: non-null
|
||||
owner = private, "the model picker only shows the endpoint to that user") and
|
||||
holds a decrypted `api_key`. start_comparison copies the matched row's api_key
|
||||
into the caller-owned [CMP] session's headers, which then drives that session's
|
||||
/api/chat_stream calls — so an UNSCOPED base_url match would let a user mint a
|
||||
comparison bound to ANOTHER user's private endpoint and spend that owner's
|
||||
api_key / reach whatever base_url they configured. Mirrors
|
||||
session_routes._owned_endpoint. A null/empty owner is a no-op (single-user /
|
||||
legacy mode).
|
||||
"""
|
||||
from core.database import ModelEndpoint
|
||||
from src.auth_helpers import owner_filter
|
||||
q = db.query(ModelEndpoint).filter(ModelEndpoint.base_url == base_url)
|
||||
return owner_filter(q, ModelEndpoint, owner).first()
|
||||
|
||||
|
||||
def _owned_endpoint_by_id(db, endpoint_id, owner):
|
||||
"""ModelEndpoint whose id == `endpoint_id` and is VISIBLE to `owner` (their
|
||||
own rows + legacy null-owner "shared" rows); None otherwise.
|
||||
|
||||
Preferred over _owned_endpoint_by_url for credential resolution: two visible
|
||||
endpoints can share the same base_url but hold DIFFERENT api_keys (e.g. two
|
||||
accounts on the same provider). A base_url-only match returns whichever row
|
||||
sorts first, so it can copy the WRONG owner-scoped key into the [CMP] session.
|
||||
An id pins the exact registered endpoint, so /api/compare/start prefers it and
|
||||
only falls back to URL matching for legacy / admin raw-URL callers. Owner
|
||||
scoping is identical to _owned_endpoint_by_url (a null/empty owner is a no-op).
|
||||
"""
|
||||
from core.database import ModelEndpoint
|
||||
from src.auth_helpers import owner_filter
|
||||
q = db.query(ModelEndpoint).filter(ModelEndpoint.id == endpoint_id)
|
||||
return owner_filter(q, ModelEndpoint, owner).first()
|
||||
|
||||
|
||||
class RecordVoteRequest(BaseModel):
|
||||
prompt: str
|
||||
models: List[str]
|
||||
winner: str # model name or "tie"
|
||||
is_blind: bool = True
|
||||
|
||||
|
||||
def setup_compare_routes(session_manager: SessionManager):
|
||||
"""Setup comparison routes."""
|
||||
|
||||
@router.post("/start")
|
||||
def start_comparison(
|
||||
request: Request,
|
||||
prompt: str = Form(...),
|
||||
model_a: str = Form(...),
|
||||
model_b: str = Form(...),
|
||||
endpoint_a: str = Form(""),
|
||||
endpoint_b: str = Form(""),
|
||||
endpoint_a_id: str = Form(""),
|
||||
endpoint_b_id: str = Form(""),
|
||||
is_blind: str = Form("true"),
|
||||
):
|
||||
"""Create two ephemeral sessions and a comparison record.
|
||||
|
||||
Returns the comparison ID and the two session IDs so the client
|
||||
can fire two independent SSE streams to /api/chat_stream.
|
||||
"""
|
||||
user = getattr(request.state, 'current_user', None)
|
||||
comp_id = str(uuid.uuid4())
|
||||
sid_a = str(uuid.uuid4())
|
||||
sid_b = str(uuid.uuid4())
|
||||
|
||||
# Blind mapping: randomly assign left/right
|
||||
blind = str(is_blind).lower() == "true"
|
||||
if blind:
|
||||
mapping = {"left": "a", "right": "b"}
|
||||
if random.random() > 0.5:
|
||||
mapping = {"left": "b", "right": "a"}
|
||||
else:
|
||||
mapping = {"left": "a", "right": "b"}
|
||||
|
||||
# Map session IDs to left/right based on blind mapping
|
||||
session_left = sid_a if mapping["left"] == "a" else sid_b
|
||||
session_right = sid_a if mapping["right"] == "a" else sid_b
|
||||
|
||||
# In blind mode, name the helper sessions by their neutral slot
|
||||
# ("Model A" / "Model B") instead of the real model. Otherwise the
|
||||
# session name leaks the model in the sidebar and GET /api/sessions,
|
||||
# de-anonymizing the comparison before the user votes (issue #1285).
|
||||
slot_name = {session_left: "Model A", session_right: "Model B"}
|
||||
|
||||
# SECURITY: resolve and validate BOTH endpoints before creating any
|
||||
# session. Compare copies a registered endpoint's Authorization header
|
||||
# into the [CMP] session, so validating one endpoint while creating its
|
||||
# session, then rejecting the other, would leave a partial compare
|
||||
# session behind with that header attached. Doing all the owner-scope
|
||||
# resolution + raw-URL rejection up front means a 403 on either endpoint
|
||||
# aborts the whole request with nothing created and no header copied.
|
||||
from src.endpoint_resolver import build_chat_url, build_headers, normalize_base
|
||||
resolved = []
|
||||
db = SessionLocal()
|
||||
try:
|
||||
for sid, model, endpoint, endpoint_id in [
|
||||
(sid_a, model_a, endpoint_a, endpoint_a_id),
|
||||
(sid_b, model_b, endpoint_b, endpoint_b_id),
|
||||
]:
|
||||
# Prefer an explicit endpoint id: it pins the EXACT registered
|
||||
# endpoint (and its api_key), even when two endpoints visible to
|
||||
# the caller share a base_url with different keys — a URL-only
|
||||
# match would copy whichever row sorts first, i.e. possibly the
|
||||
# wrong key. Fall back to URL resolution only for legacy / admin
|
||||
# raw-URL callers that don't send an id.
|
||||
eid = endpoint_id.strip() if isinstance(endpoint_id, str) else ""
|
||||
if eid:
|
||||
ep = _owned_endpoint_by_id(db, eid, user)
|
||||
if ep is None:
|
||||
# An id the caller can't see (wrong owner / deleted) must
|
||||
# NOT silently fall back to a same-URL row with a different
|
||||
# key — that's exactly the mix-up ids exist to prevent.
|
||||
raise HTTPException(404, "Model endpoint not found")
|
||||
# The id already resolved the endpoint; ignore any raw URL the
|
||||
# caller also sent and dial the stored config instead.
|
||||
endpoint = ep.base_url
|
||||
elif not endpoint:
|
||||
raise HTTPException(
|
||||
422, "endpoint_a/endpoint_b or endpoint_a_id/endpoint_b_id is required"
|
||||
)
|
||||
else:
|
||||
# Resolve the supplied URL to a ModelEndpoint the caller owns
|
||||
# (their own rows + legacy null-owner shared rows), scoped so a
|
||||
# comparison can't borrow another user's private endpoint key.
|
||||
base = normalize_base(endpoint)
|
||||
ep = _owned_endpoint_by_url(db, base, user)
|
||||
# Reject *unregistered* raw URLs for signed-in non-admins; a
|
||||
# matched registered endpoint supplies an id so the caller can
|
||||
# still compare endpoints they own. Blanket-rejecting here (the
|
||||
# earlier `endpoint_id=None` call) locked non-admins out of
|
||||
# compare entirely, since compare resolves endpoints by URL with
|
||||
# no endpoint_id. Mirrors the gallery inpaint/harmonize checks.
|
||||
# Raised here (phase 1), before any session exists.
|
||||
_reject_raw_endpoint_url_for_non_admin(
|
||||
request, user, str(ep.id) if ep is not None else None, endpoint
|
||||
)
|
||||
# Bind the [CMP] session to the RESOLVED endpoint, not the raw
|
||||
# caller-supplied string. When the URL matches a registered
|
||||
# endpoint visible to the caller, use that row's own normalized
|
||||
# base URL (the same value owner scoping + endpoint validation
|
||||
# already vetted) so the session dials exactly where the stored
|
||||
# config points. The raw `endpoint` only survives for callers
|
||||
# allowed to pass one — admins / single-user mode, where
|
||||
# `_reject_raw_endpoint_url_for_non_admin` is a no-op and `ep`
|
||||
# is None. Mirrors the registered-endpoint path in session_routes.
|
||||
session_endpoint_url = (
|
||||
build_chat_url(normalize_base(ep.base_url)) if ep is not None else endpoint
|
||||
)
|
||||
# Headers come only from a matched endpoint's key; None when
|
||||
# `ep` is None (raw admin URL or no match), so a comparison can
|
||||
# never inherit another user's key/headers.
|
||||
headers = build_headers(ep.api_key, ep.base_url) if (ep and ep.api_key) else None
|
||||
resolved.append((sid, model, session_endpoint_url, headers))
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
# Both endpoints validated — only now create the ephemeral [CMP]
|
||||
# sessions and copy any resolved headers.
|
||||
for sid, model, session_endpoint_url, headers in resolved:
|
||||
name = f"[CMP] {slot_name[sid]}" if blind else f"[CMP] {model.split('/')[-1]}"
|
||||
session_manager.create_session(
|
||||
session_id=sid,
|
||||
name=name,
|
||||
endpoint_url=session_endpoint_url,
|
||||
model=model,
|
||||
rag=False,
|
||||
owner=user,
|
||||
)
|
||||
if headers:
|
||||
s = session_manager.sessions.get(sid)
|
||||
if s:
|
||||
s.headers = headers
|
||||
|
||||
# Store comparison record
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = Comparison(
|
||||
id=comp_id,
|
||||
prompt=prompt,
|
||||
model_a=model_a,
|
||||
model_b=model_b,
|
||||
# Record the URL the session actually dials. For URL callers this
|
||||
# is their raw input; for id-only callers (empty endpoint_a/_b)
|
||||
# fall back to the resolved endpoint URL so the column stays
|
||||
# meaningful and non-null. resolved is in [a, b] order.
|
||||
endpoint_a=endpoint_a or resolved[0][2],
|
||||
endpoint_b=endpoint_b or resolved[1][2],
|
||||
is_blind=blind,
|
||||
blind_mapping=json.dumps(mapping),
|
||||
owner=user,
|
||||
)
|
||||
db.add(comp)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
# In blind mode, withhold the model identities AND the left/right
|
||||
# mapping from the response. The client already knows model_a/model_b
|
||||
# (it sent them), so returning either would defeat blind mode. They are
|
||||
# revealed by POST /api/compare/{id}/vote once the user has voted (#1285).
|
||||
return {
|
||||
"id": comp_id,
|
||||
"session_left": session_left,
|
||||
"session_right": session_right,
|
||||
"model_left": None if blind else (model_a if mapping["left"] == "a" else model_b),
|
||||
"model_right": None if blind else (model_a if mapping["right"] == "a" else model_b),
|
||||
"is_blind": blind,
|
||||
"mapping": None if blind else mapping,
|
||||
}
|
||||
|
||||
@router.post("/{comp_id}/vote")
|
||||
def vote_comparison(
|
||||
request: Request,
|
||||
comp_id: str,
|
||||
winner: str = Form(...), # "left", "right", or "tie"
|
||||
):
|
||||
"""Record the user's vote and reveal model names if blind."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = db.query(Comparison).filter(Comparison.id == comp_id).first()
|
||||
if not comp:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
# SECURITY: strict ownership — null-owner Comparisons were
|
||||
# accessible to every user.
|
||||
if user and comp.owner != user:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
if comp.winner:
|
||||
raise HTTPException(400, "Already voted")
|
||||
|
||||
mapping = json.loads(comp.blind_mapping) if comp.blind_mapping else {"left": "a", "right": "b"}
|
||||
|
||||
if winner == "tie":
|
||||
comp.winner = "tie"
|
||||
elif winner == "left":
|
||||
comp.winner = mapping["left"]
|
||||
elif winner == "right":
|
||||
comp.winner = mapping["right"]
|
||||
else:
|
||||
raise HTTPException(400, "winner must be 'left', 'right', or 'tie'")
|
||||
|
||||
comp.voted_at = datetime.utcnow()
|
||||
db.commit()
|
||||
|
||||
return {
|
||||
"winner": comp.winner,
|
||||
"model_a": comp.model_a,
|
||||
"model_b": comp.model_b,
|
||||
"revealed": {
|
||||
"left": comp.model_a if mapping["left"] == "a" else comp.model_b,
|
||||
"right": comp.model_a if mapping["right"] == "a" else comp.model_b,
|
||||
},
|
||||
}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.post("/record")
|
||||
def record_comparison(request: Request, body: RecordVoteRequest):
|
||||
"""Lightweight endpoint to record a comparison vote from the frontend."""
|
||||
user = get_current_user(request)
|
||||
comp_id = str(uuid.uuid4())
|
||||
|
||||
model_a = body.models[0] if len(body.models) > 0 else ""
|
||||
model_b = body.models[1] if len(body.models) > 1 else ""
|
||||
|
||||
# For N>2 models, store the full list as JSON in blind_mapping
|
||||
if len(body.models) > 2:
|
||||
blind_mapping = json.dumps({"models": body.models})
|
||||
else:
|
||||
blind_mapping = None
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = Comparison(
|
||||
id=comp_id,
|
||||
prompt=body.prompt[:500],
|
||||
model_a=model_a,
|
||||
model_b=model_b,
|
||||
endpoint_a="",
|
||||
endpoint_b="",
|
||||
winner=body.winner,
|
||||
is_blind=body.is_blind,
|
||||
blind_mapping=blind_mapping,
|
||||
voted_at=datetime.utcnow(),
|
||||
owner=user,
|
||||
)
|
||||
db.add(comp)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return {"status": "ok", "id": comp_id}
|
||||
|
||||
@router.get("/history")
|
||||
def list_comparisons(request: Request):
|
||||
"""List past comparisons."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
q = db.query(Comparison)
|
||||
if user:
|
||||
q = q.filter(Comparison.owner == user)
|
||||
comps = q.order_by(Comparison.created_at.desc()).limit(50).all()
|
||||
return [
|
||||
{
|
||||
"id": c.id,
|
||||
"prompt": c.prompt[:100],
|
||||
"model_a": c.model_a,
|
||||
"model_b": c.model_b,
|
||||
"winner": c.winner,
|
||||
"is_blind": c.is_blind,
|
||||
"voted_at": c.voted_at.isoformat() if c.voted_at else None,
|
||||
"created_at": c.created_at.isoformat() if c.created_at else None,
|
||||
}
|
||||
for c in comps
|
||||
]
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@router.delete("/{comp_id}")
|
||||
def delete_comparison(request: Request, comp_id: str):
|
||||
"""Delete a comparison and its ephemeral sessions."""
|
||||
user = get_current_user(request)
|
||||
db = SessionLocal()
|
||||
try:
|
||||
comp = db.query(Comparison).filter(Comparison.id == comp_id).first()
|
||||
if not comp:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
# SECURITY: strict ownership — null-owner Comparisons were
|
||||
# accessible to every user.
|
||||
if user and comp.owner != user:
|
||||
raise HTTPException(404, "Comparison not found")
|
||||
db.delete(comp)
|
||||
db.commit()
|
||||
return {"status": "deleted"}
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
return router
|
||||
_sys.modules[__name__] = _canonical
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user