# Production-like compose. Requires a populated `.env` next to this # file: at minimum POSTGRES_PASSWORD must be set to a non-default # value (the `?required` form below fails fast otherwise). The # frontend container expects HTTPS in front (Caddy/Traefik/nginx) # because COOKIE_SECURE=true browsers will refuse to send the session # cookie over plain HTTP. services: postgres: image: postgres:16-alpine environment: POSTGRES_USER: ${POSTGRES_USER:-mangalord} POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?POSTGRES_PASSWORD must be set in .env} POSTGRES_DB: ${POSTGRES_DB:-mangalord} volumes: - postgres-data:/var/lib/postgresql/data healthcheck: test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER:-mangalord}"] interval: 5s timeout: 5s retries: 10 # Without this, a single Postgres crash leaves backend's # `depends_on: service_healthy` blocking every future restart and # takes the whole stack offline. Match the policy already set on # tor / docker-socket-proxy / vision-manager. restart: unless-stopped tor: # SOCKS5 proxy for the crawler, plus a control port so the backend # can signal NEWNYM on bad pages. See tor/torrc for the daemon # config; both ports are only `expose`d (compose-internal), never # bound on the host. # # We bypass dockurr/tor's stock entrypoint because it binds the # control port to localhost (unreachable from the backend # container) and skips its own HashedControlPassword injection # when the user's torrc declares a ControlPort. Our wrapper # (tor/entrypoint.sh) generates the hash from $PASSWORD and execs # tor with our torrc. Backend authenticates with the same plain # string via CRAWLER_TOR_CONTROL_PASSWORD. image: dockurr/tor:latest entrypoint: ["/bin/sh", "/usr/local/bin/mangalord-entrypoint.sh"] environment: PASSWORD: ${TOR_CONTROL_PASSWORD:?TOR_CONTROL_PASSWORD must be set in .env} volumes: - ./tor/torrc:/etc/tor/torrc:ro - ./tor/entrypoint.sh:/usr/local/bin/mangalord-entrypoint.sh:ro expose: - "9050" - "9051" # Wait for both control + SOCKS ports to listen before downstream # services start. dockurr/tor's main process spawns before tor # itself is bound, so `service_started` alone races the first # NEWNYM call. healthcheck: test: ["CMD-SHELL", "nc -z 127.0.0.1 9050 && nc -z 127.0.0.1 9051"] interval: 5s timeout: 5s retries: 20 start_period: 30s restart: unless-stopped backend: build: ./backend depends_on: postgres: condition: service_healthy tor: condition: service_healthy environment: DATABASE_URL: postgres://${POSTGRES_USER:-mangalord}:${POSTGRES_PASSWORD:?POSTGRES_PASSWORD must be set in .env}@postgres:5432/${POSTGRES_DB:-mangalord} BIND_ADDRESS: 0.0.0.0:8080 STORAGE_DIR: /var/lib/mangalord/storage RUST_LOG: ${RUST_LOG:-info,mangalord=debug} # Postgres connection-pool sizing — see .env.example for context. DB_MAX_CONNECTIONS: ${DB_MAX_CONNECTIONS:-20} DB_ACQUIRE_TIMEOUT_SECS: ${DB_ACQUIRE_TIMEOUT_SECS:-10} # Auth / cookies — see .env.example for context. COOKIE_SECURE: ${COOKIE_SECURE:-true} COOKIE_DOMAIN: ${COOKIE_DOMAIN:-} SESSION_TTL_DAYS: ${SESSION_TTL_DAYS:-30} AUTH_RATE_PER_SEC: ${AUTH_RATE_PER_SEC:-5} AUTH_RATE_BURST: ${AUTH_RATE_BURST:-10} # The SvelteKit container is the single trusted hop in front of the # backend and stamps the real client IP, so per-IP rate limiting is on # by default here (unlike the backend's safe-by-default `false`). AUTH_TRUSTED_PROXY: ${AUTH_TRUSTED_PROXY:-true} # CORS — same-origin by default; populate when serving the API on # a different host than the frontend. CORS_ALLOWED_ORIGINS: ${CORS_ALLOWED_ORIGINS:-} # Admin CSRF allowlist. Empty in default deploy → cookie-auth admin # mutations are refused (fail-closed since 0.87.2). Set to the # SvelteKit origin (e.g. https://app.example.com) for browser deploys. ADMIN_ALLOWED_ORIGINS: ${ADMIN_ALLOWED_ORIGINS:-} # Admin bootstrap. Half-set is treated as "skip bootstrap"; either # both or neither. Set on first boot, then unset for subsequent # deploys so the password isn't re-applied accidentally (existing # users' passwords are never overwritten by this). ADMIN_USERNAME: ${ADMIN_USERNAME:-} ADMIN_PASSWORD: ${ADMIN_PASSWORD:-} # Site-wide auth gates (env-only; never persisted to app_settings). PRIVATE_MODE: ${PRIVATE_MODE:-false} ALLOW_SELF_REGISTER: ${ALLOW_SELF_REGISTER:-true} # Upload limits. MAX_REQUEST_BYTES: ${MAX_REQUEST_BYTES:-209715200} MAX_FILE_BYTES: ${MAX_FILE_BYTES:-20971520} MAX_PAGES_PER_CHAPTER: ${MAX_PAGES_PER_CHAPTER:-2000} # Crawler boot seeds (first-boot only; switch to dashboard-editable # once the app_settings row exists). CRAWLER_START_URL: ${CRAWLER_START_URL:-} CRAWLER_CDN_HOST: ${CRAWLER_CDN_HOST:-} CRAWLER_LIMIT: ${CRAWLER_LIMIT:-0} CRAWLER_JOB_TIMEOUT_SECS: ${CRAWLER_JOB_TIMEOUT_SECS:-600} CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES: ${CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES:-10} CRAWLER_BROWSER_RESTART_THRESHOLD: ${CRAWLER_BROWSER_RESTART_THRESHOLD:-3} CRAWLER_SSRF_INTERCEPT: ${CRAWLER_SSRF_INTERCEPT:-true} # Crawler daemon schedule + retention. CRAWLER_DAEMON=false keeps # the in-process scheduler off; the dashboard force-resync still works. CRAWLER_DAEMON: ${CRAWLER_DAEMON:-true} CRAWLER_DAILY_AT: ${CRAWLER_DAILY_AT:-00:00} CRAWLER_TZ: ${CRAWLER_TZ:-UTC} CRAWLER_IDLE_TIMEOUT_S: ${CRAWLER_IDLE_TIMEOUT_S:-600} CRAWLER_CHAPTER_WORKERS: ${CRAWLER_CHAPTER_WORKERS:-1} CRAWLER_JOB_RETENTION_DAYS: ${CRAWLER_JOB_RETENTION_DAYS:-7} CRAWL_METRICS_RETENTION_DAYS: ${CRAWL_METRICS_RETENTION_DAYS:-90} # Crawler politeness — pause between source / CDN fetches (ms). CRAWLER_RATE_MS: ${CRAWLER_RATE_MS:-1000} CRAWLER_CDN_RATE_MS: ${CRAWLER_CDN_RATE_MS:-1000} # Crawler session/identity to the source (cookies, UA). CRAWLER_USER_AGENT: ${CRAWLER_USER_AGENT:-} CRAWLER_PHPSESSID: ${CRAWLER_PHPSESSID:-} CRAWLER_COOKIE_DOMAIN: ${CRAWLER_COOKIE_DOMAIN:-} # Crawler download safety — env-only, never persisted. CRAWLER_DOWNLOAD_ALLOWLIST: ${CRAWLER_DOWNLOAD_ALLOWLIST:-} CRAWLER_ALLOW_ANY_HOST: ${CRAWLER_ALLOW_ANY_HOST:-false} CRAWLER_MAX_IMAGE_BYTES: ${CRAWLER_MAX_IMAGE_BYTES:-33554432} CRAWLER_MAX_IMAGES_PER_CHAPTER: ${CRAWLER_MAX_IMAGES_PER_CHAPTER:-2000} # System-chromium override for the crawler. Leave blank to use the # bundled fetcher; set to e.g. /usr/bin/chromium-headless-shell on # arm64 deployments. Pair with `--build-arg INSTALL_CHROMIUM=true` # so the image actually contains the binary. CRAWLER_CHROMIUM_BINARY: ${CRAWLER_CHROMIUM_BINARY:-} # TOR proxy + NEWNYM recircuit (see .env.example for details). # Defaults assume the bundled `tor` service above; override # CRAWLER_PROXY= and CRAWLER_TOR_CONTROL_URL= (both empty) in # .env to disable. CRAWLER_TOR_CONTROL_PASSWORD MUST match the # tor service's PASSWORD (both wired to the same TOR_CONTROL_PASSWORD # .env var below). CRAWLER_PROXY: ${CRAWLER_PROXY-socks5h://tor:9050} CRAWLER_TOR_CONTROL_URL: ${CRAWLER_TOR_CONTROL_URL-tcp://tor:9051} CRAWLER_TOR_CONTROL_PASSWORD: ${TOR_CONTROL_PASSWORD:?TOR_CONTROL_PASSWORD must be set in .env} CRAWLER_TOR_RECIRCUIT_MAX_ATTEMPTS: ${CRAWLER_TOR_RECIRCUIT_MAX_ATTEMPTS:-3} # Tor cookie-file auth alternative — the controller prefers cookie # when both are present. Leave unset for the bundled # HashedControlPassword path above. CRAWLER_TOR_CONTROL_COOKIE_PATH: ${CRAWLER_TOR_CONTROL_COOKIE_PATH:-} # Analysis worker — boot seeds + env-only API key. ANALYSIS_ENABLED: ${ANALYSIS_ENABLED:-false} # Engine selection (deploy-time): `ocr` (in-process ocrs, default). # The `vision` (local LLM) backend is temporarily disabled — the worker # forces OCR regardless, so a `vision` override here is currently inert. # Model paths default to the image-baked /models. ANALYSIS_BACKEND: ${ANALYSIS_BACKEND:-ocr} OCRS_DETECTION_MODEL: ${OCRS_DETECTION_MODEL:-/models/text-detection.rten} OCRS_RECOGNITION_MODEL: ${OCRS_RECOGNITION_MODEL:-/models/text-recognition.rten} ANALYSIS_VISION_URL: ${ANALYSIS_VISION_URL:-} ANALYSIS_VISION_MODEL: ${ANALYSIS_VISION_MODEL:-} ANALYSIS_WORKERS: ${ANALYSIS_WORKERS:-1} ANALYSIS_JOB_TIMEOUT_SECS: ${ANALYSIS_JOB_TIMEOUT_SECS:-600} # API key never gets persisted to app_settings; bearer-attached to # every vision call. ANALYSIS_API_KEY: ${ANALYSIS_API_KEY:-} # Analysis tuning knobs — boot seeds, then live-editable in dashboard. ANALYSIS_MAX_TOKENS: ${ANALYSIS_MAX_TOKENS:-4096} ANALYSIS_MAX_PIXELS: ${ANALYSIS_MAX_PIXELS:-1000000} ANALYSIS_MIN_SLICE_HEIGHT: ${ANALYSIS_MIN_SLICE_HEIGHT:-640} ANALYSIS_SLICE_OVERLAP: ${ANALYSIS_SLICE_OVERLAP:-0.12} ANALYSIS_TALL_ASPECT: ${ANALYSIS_TALL_ASPECT:-1.6} ANALYSIS_MAX_SLICES: ${ANALYSIS_MAX_SLICES:-16} ANALYSIS_MAX_IMAGE_BYTES: ${ANALYSIS_MAX_IMAGE_BYTES:-8388608} ANALYSIS_OCR_MAX_DECODE_PIXELS: ${ANALYSIS_OCR_MAX_DECODE_PIXELS:-100000000} ANALYSIS_RESPONSE_FORMAT: ${ANALYSIS_RESPONSE_FORMAT:-json_schema} ANALYSIS_FREQUENCY_PENALTY: ${ANALYSIS_FREQUENCY_PENALTY:-0.3} ANALYSIS_TEMPERATURE: ${ANALYSIS_TEMPERATURE:-0.0} # Vision readiness gate (env-ONLY, not a dashboard setting). When set, # the analysis worker refuses to lease a page until this answers 2xx, so # the vision-manager autoscaler can idle-stop the vision container # without jobs burning their retries / landing `failed` rows. Leave # empty (the default) to disable the gate for an always-on endpoint. # Pair with the `ai` profile services below. See VISION-AUTOSCALE.md. ANALYSIS_VISION_HEALTH_URL: ${ANALYSIS_VISION_HEALTH_URL:-} volumes: - storage-data:/var/lib/mangalord/storage # No host port mapping in the default setup — the frontend proxies # /api/* through its hooks.server.ts. Expose :8080 only if you want # to hit the API directly from the host (e.g., bot scripts during # development). expose: - "8080" restart: unless-stopped frontend: build: ./frontend depends_on: - backend environment: # SvelteKit's hooks.server.ts proxies /api/* to this URL so the # browser only ever talks to :3000 and cookies stay same-origin. # `.env` override exists for deploys where the SvelteKit container # needs to reach a backend on a different hostname (typical with # an external reverse proxy in front of axum). BACKEND_URL: ${BACKEND_URL:-http://backend:8080} # Per-request wall-clock cap for the /api/* reverse proxy. Defaults # to 300000 (5 min) in hooks.server.ts; raise/lower via .env. BACKEND_PROXY_TIMEOUT_MS: ${BACKEND_PROXY_TIMEOUT_MS:-300000} ports: - "3000:3000" restart: unless-stopped # ----- Vision autoscaling (profile: ai) ----------------------------------- # Two extra containers that idle-stop the mangalord-vision (llama.cpp) # container when there is no analysis work and start it back up on demand. # Gated behind `profiles: [ai]` so a vanilla `docker compose up` is # unaffected — bring them up with `docker compose --profile ai up -d`. # The vision container itself is NOT defined here (it lives elsewhere on the # host); the manager drives it by name. See VISION-AUTOSCALE.md. # Scoped Docker access for the manager. The proxy gates by API *section* # (not per-method), so CONTAINERS=1 + POST=1 permits the full /containers # lifecycle — inspect/start/stop, but also create/kill/restart/update/ # rename/remove. It does NOT expose exec, images, volumes, networks, swarm, # etc. (all default-denied). The trust boundary is therefore: (a) this is an # internal-only network reachable solely by vision-manager, and (b) the raw # host socket is mounted HERE and nowhere else — never on the backend. A # backend RCE still cannot reach the Docker API. If you need true start/stop- # only granularity, front the socket with an allow-list reverse proxy instead. docker-socket-proxy: image: tecnativa/docker-socket-proxy:latest profiles: ["ai"] environment: CONTAINERS: 1 # allow the /containers/* section POST: 1 # allow write methods (start/stop are POSTs) # Everything else stays at its default-deny (EXEC, IMAGES, NETWORKS, ...). volumes: - /var/run/docker.sock:/var/run/docker.sock:ro networks: - vision-internal restart: unless-stopped vision-manager: build: ./vision-manager profiles: ["ai"] depends_on: postgres: condition: service_healthy docker-socket-proxy: condition: service_started environment: # Read-only role — apply vision-manager/readonly-role.sql once, then set # VISION_MANAGER_DATABASE_URL in .env. Do NOT reuse the backend creds. DATABASE_URL: ${VISION_MANAGER_DATABASE_URL:?set VISION_MANAGER_DATABASE_URL in .env (vision_manager read-only role)} VISION_CONTAINER: ${VISION_CONTAINER:-mangalord-vision} VISION_HEALTH_URL: ${VISION_HEALTH_URL:-http://mangalord-vision:8000/health} DOCKER_HOST: tcp://docker-socket-proxy:2375 POLL_INTERVAL: ${VISION_POLL_INTERVAL:-20} STOP_DEBOUNCE: ${VISION_STOP_DEBOUNCE:-600} START_HEALTH_TIMEOUT: ${VISION_START_HEALTH_TIMEOUT:-300} RESPECT_CRAWL_MUTEX: ${VISION_RESPECT_CRAWL_MUTEX:-1} MAX_UPTIME: ${VISION_MAX_UPTIME:-0} # Memory-pressure yield — stop vision when the HOST is short on RAM so a # spike elsewhere can finish without the kernel OOM-killer. See # VISION-MEMORY-YIELD.md. MEM_YIELD_ENABLED: ${VISION_MEM_YIELD_ENABLED:-1} MEM_HIGH_WATERMARK_PCT: ${VISION_MEM_HIGH_WATERMARK_PCT:-92} MEM_LOW_WATERMARK_PCT: ${VISION_MEM_LOW_WATERMARK_PCT:-80} MEM_YIELD_COOLDOWN: ${VISION_MEM_YIELD_COOLDOWN:-300} MEM_POLL_INTERVAL: ${VISION_MEM_POLL_INTERVAL:-5} networks: - default # reach postgres + the vision container by name - vision-internal # reach the socket-proxy restart: unless-stopped networks: default: # Internal-only: no route to the outside world. Only the manager and the # socket-proxy sit on it, so nothing else can reach the Docker API. vision-internal: internal: true volumes: postgres-data: storage-data: