mistral-nemo-large holds ~9.2 GB of the 11 GB card it shares with Speaches, which starves Whisper and breaks voice transcription. gemma4:e2b holds 1.9 GB and is faster. The deployed stack already overrides this via OLLAMA_AGENT_MODEL; this aligns the default so a deployment without that override does not reintroduce the contention. Co-Authored-By: Claude <noreply@anthropic.com>
35 lines
767 B
Bash
35 lines
767 B
Bash
# Webber Configuration
|
|
# Copy to .env and customize
|
|
|
|
# Application
|
|
DEBUG=true
|
|
LOG_LEVEL=DEBUG
|
|
|
|
# Server
|
|
HOST=0.0.0.0
|
|
PORT=8086
|
|
|
|
# CORS (comma-separated)
|
|
CORS_ORIGINS=["http://localhost:3000","http://localhost:8080"]
|
|
|
|
# LLM - Ollama (tower-of-joy)
|
|
OLLAMA_URL=http://192.168.86.149:11434
|
|
OLLAMA_AGENT_MODEL=gemma4:e2b
|
|
OLLAMA_EMBED_MODEL=nomic-embed-text:latest
|
|
|
|
# Auth - Tatlock integration (optional)
|
|
# TATLOCK_API_URL=http://tatlock:8000
|
|
# INTERNAL_API_KEY=your-internal-key
|
|
|
|
# Web search - SearXNG (container name on docker-dataplane; internal port 8080)
|
|
# SEARXNG_URL=http://searxng:8080
|
|
|
|
# Tool execution
|
|
TOOL_TIMEOUT_SECONDS=120
|
|
SANDBOX_ENABLED=true
|
|
# ALLOWED_PATHS=["/home/user/projects","/tmp/webber"]
|
|
|
|
# Sessions
|
|
SESSION_TTL_HOURS=24
|
|
MAX_CONTEXT_TOKENS=128000
|