The chapter upload handler read every `page` part fully into a Vec before
writing any, so peak memory was the whole chapter (bounded only by the
200 MiB body limit and amplified by concurrent uploads). It also accepted
an unbounded number of pages.
Stream each page part to a `staging/{upload_id}/…` key as it arrives — at
most one page's bytes are held at a time — then, once the chapter row (and
its id) exists, promote each staged blob to its final key via a new
`Storage::rename` (LocalStorage: fs rename; default impl: stream+delete for
future backends). Finalization is all-or-nothing: on any failure the DB
rolls back and both staged and already-finalized blobs are cleaned up.
Add MAX_PAGES_PER_CHAPTER (UploadConfig, default 2000, 0 = disabled),
rejecting an over-cap upload with 413 before any DB write. Also document
the crawler-side CRAWLER_MAX_IMAGES_PER_CHAPTER (added earlier) in
.env.example + docker-compose so the env-coverage test passes.
Tests: LocalStorage rename unit tests; a 413 over-cap upload test; existing
rollback + happy-path upload tests still green (the fault-injecting storage
counts put/put_stream, so mid-upload failure still rolls back).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
284 lines
14 KiB
YAML
284 lines
14 KiB
YAML
# Production-like compose. Requires a populated `.env` next to this
|
|
# file: at minimum POSTGRES_PASSWORD must be set to a non-default
|
|
# value (the `?required` form below fails fast otherwise). The
|
|
# frontend container expects HTTPS in front (Caddy/Traefik/nginx)
|
|
# because COOKIE_SECURE=true browsers will refuse to send the session
|
|
# cookie over plain HTTP.
|
|
services:
|
|
postgres:
|
|
image: postgres:16-alpine
|
|
environment:
|
|
POSTGRES_USER: ${POSTGRES_USER:-mangalord}
|
|
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?POSTGRES_PASSWORD must be set in .env}
|
|
POSTGRES_DB: ${POSTGRES_DB:-mangalord}
|
|
volumes:
|
|
- postgres-data:/var/lib/postgresql/data
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER:-mangalord}"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
# Without this, a single Postgres crash leaves backend's
|
|
# `depends_on: service_healthy` blocking every future restart and
|
|
# takes the whole stack offline. Match the policy already set on
|
|
# tor / docker-socket-proxy / vision-manager.
|
|
restart: unless-stopped
|
|
|
|
tor:
|
|
# SOCKS5 proxy for the crawler, plus a control port so the backend
|
|
# can signal NEWNYM on bad pages. See tor/torrc for the daemon
|
|
# config; both ports are only `expose`d (compose-internal), never
|
|
# bound on the host.
|
|
#
|
|
# We bypass dockurr/tor's stock entrypoint because it binds the
|
|
# control port to localhost (unreachable from the backend
|
|
# container) and skips its own HashedControlPassword injection
|
|
# when the user's torrc declares a ControlPort. Our wrapper
|
|
# (tor/entrypoint.sh) generates the hash from $PASSWORD and execs
|
|
# tor with our torrc. Backend authenticates with the same plain
|
|
# string via CRAWLER_TOR_CONTROL_PASSWORD.
|
|
image: dockurr/tor:latest
|
|
entrypoint: ["/bin/sh", "/usr/local/bin/mangalord-entrypoint.sh"]
|
|
environment:
|
|
PASSWORD: ${TOR_CONTROL_PASSWORD:?TOR_CONTROL_PASSWORD must be set in .env}
|
|
volumes:
|
|
- ./tor/torrc:/etc/tor/torrc:ro
|
|
- ./tor/entrypoint.sh:/usr/local/bin/mangalord-entrypoint.sh:ro
|
|
expose:
|
|
- "9050"
|
|
- "9051"
|
|
# Wait for both control + SOCKS ports to listen before downstream
|
|
# services start. dockurr/tor's main process spawns before tor
|
|
# itself is bound, so `service_started` alone races the first
|
|
# NEWNYM call.
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "nc -z 127.0.0.1 9050 && nc -z 127.0.0.1 9051"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 20
|
|
start_period: 30s
|
|
restart: unless-stopped
|
|
|
|
backend:
|
|
build: ./backend
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
tor:
|
|
condition: service_healthy
|
|
environment:
|
|
DATABASE_URL: postgres://${POSTGRES_USER:-mangalord}:${POSTGRES_PASSWORD:?POSTGRES_PASSWORD must be set in .env}@postgres:5432/${POSTGRES_DB:-mangalord}
|
|
BIND_ADDRESS: 0.0.0.0:8080
|
|
STORAGE_DIR: /var/lib/mangalord/storage
|
|
RUST_LOG: ${RUST_LOG:-info,mangalord=debug}
|
|
# Auth / cookies — see .env.example for context.
|
|
COOKIE_SECURE: ${COOKIE_SECURE:-true}
|
|
COOKIE_DOMAIN: ${COOKIE_DOMAIN:-}
|
|
SESSION_TTL_DAYS: ${SESSION_TTL_DAYS:-30}
|
|
AUTH_RATE_PER_SEC: ${AUTH_RATE_PER_SEC:-5}
|
|
AUTH_RATE_BURST: ${AUTH_RATE_BURST:-10}
|
|
# CORS — same-origin by default; populate when serving the API on
|
|
# a different host than the frontend.
|
|
CORS_ALLOWED_ORIGINS: ${CORS_ALLOWED_ORIGINS:-}
|
|
# Admin CSRF allowlist. Empty in default deploy → cookie-auth admin
|
|
# mutations are refused (fail-closed since 0.87.2). Set to the
|
|
# SvelteKit origin (e.g. https://app.example.com) for browser deploys.
|
|
ADMIN_ALLOWED_ORIGINS: ${ADMIN_ALLOWED_ORIGINS:-}
|
|
# Admin bootstrap. Half-set is treated as "skip bootstrap"; either
|
|
# both or neither. Set on first boot, then unset for subsequent
|
|
# deploys so the password isn't re-applied accidentally (existing
|
|
# users' passwords are never overwritten by this).
|
|
ADMIN_USERNAME: ${ADMIN_USERNAME:-}
|
|
ADMIN_PASSWORD: ${ADMIN_PASSWORD:-}
|
|
# Site-wide auth gates (env-only; never persisted to app_settings).
|
|
PRIVATE_MODE: ${PRIVATE_MODE:-false}
|
|
ALLOW_SELF_REGISTER: ${ALLOW_SELF_REGISTER:-true}
|
|
# Upload limits.
|
|
MAX_REQUEST_BYTES: ${MAX_REQUEST_BYTES:-209715200}
|
|
MAX_FILE_BYTES: ${MAX_FILE_BYTES:-20971520}
|
|
MAX_PAGES_PER_CHAPTER: ${MAX_PAGES_PER_CHAPTER:-2000}
|
|
# Crawler boot seeds (first-boot only; switch to dashboard-editable
|
|
# once the app_settings row exists).
|
|
CRAWLER_START_URL: ${CRAWLER_START_URL:-}
|
|
CRAWLER_CDN_HOST: ${CRAWLER_CDN_HOST:-}
|
|
CRAWLER_LIMIT: ${CRAWLER_LIMIT:-0}
|
|
CRAWLER_JOB_TIMEOUT_SECS: ${CRAWLER_JOB_TIMEOUT_SECS:-600}
|
|
CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES: ${CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES:-10}
|
|
CRAWLER_BROWSER_RESTART_THRESHOLD: ${CRAWLER_BROWSER_RESTART_THRESHOLD:-3}
|
|
# Crawler daemon schedule + retention. CRAWLER_DAEMON=false keeps
|
|
# the in-process scheduler off; the dashboard force-resync still works.
|
|
CRAWLER_DAEMON: ${CRAWLER_DAEMON:-true}
|
|
CRAWLER_DAILY_AT: ${CRAWLER_DAILY_AT:-00:00}
|
|
CRAWLER_TZ: ${CRAWLER_TZ:-UTC}
|
|
CRAWLER_IDLE_TIMEOUT_S: ${CRAWLER_IDLE_TIMEOUT_S:-600}
|
|
CRAWLER_CHAPTER_WORKERS: ${CRAWLER_CHAPTER_WORKERS:-1}
|
|
CRAWLER_JOB_RETENTION_DAYS: ${CRAWLER_JOB_RETENTION_DAYS:-7}
|
|
CRAWL_METRICS_RETENTION_DAYS: ${CRAWL_METRICS_RETENTION_DAYS:-90}
|
|
# Crawler politeness — pause between source / CDN fetches (ms).
|
|
CRAWLER_RATE_MS: ${CRAWLER_RATE_MS:-1000}
|
|
CRAWLER_CDN_RATE_MS: ${CRAWLER_CDN_RATE_MS:-1000}
|
|
# Crawler session/identity to the source (cookies, UA).
|
|
CRAWLER_USER_AGENT: ${CRAWLER_USER_AGENT:-}
|
|
CRAWLER_PHPSESSID: ${CRAWLER_PHPSESSID:-}
|
|
CRAWLER_COOKIE_DOMAIN: ${CRAWLER_COOKIE_DOMAIN:-}
|
|
# Crawler download safety — env-only, never persisted.
|
|
CRAWLER_DOWNLOAD_ALLOWLIST: ${CRAWLER_DOWNLOAD_ALLOWLIST:-}
|
|
CRAWLER_ALLOW_ANY_HOST: ${CRAWLER_ALLOW_ANY_HOST:-false}
|
|
CRAWLER_MAX_IMAGE_BYTES: ${CRAWLER_MAX_IMAGE_BYTES:-33554432}
|
|
CRAWLER_MAX_IMAGES_PER_CHAPTER: ${CRAWLER_MAX_IMAGES_PER_CHAPTER:-2000}
|
|
# System-chromium override for the crawler. Leave blank to use the
|
|
# bundled fetcher; set to e.g. /usr/bin/chromium-headless-shell on
|
|
# arm64 deployments. Pair with `--build-arg INSTALL_CHROMIUM=true`
|
|
# so the image actually contains the binary.
|
|
CRAWLER_CHROMIUM_BINARY: ${CRAWLER_CHROMIUM_BINARY:-}
|
|
# TOR proxy + NEWNYM recircuit (see .env.example for details).
|
|
# Defaults assume the bundled `tor` service above; override
|
|
# CRAWLER_PROXY= and CRAWLER_TOR_CONTROL_URL= (both empty) in
|
|
# .env to disable. CRAWLER_TOR_CONTROL_PASSWORD MUST match the
|
|
# tor service's PASSWORD (both wired to the same TOR_CONTROL_PASSWORD
|
|
# .env var below).
|
|
CRAWLER_PROXY: ${CRAWLER_PROXY-socks5h://tor:9050}
|
|
CRAWLER_TOR_CONTROL_URL: ${CRAWLER_TOR_CONTROL_URL-tcp://tor:9051}
|
|
CRAWLER_TOR_CONTROL_PASSWORD: ${TOR_CONTROL_PASSWORD:?TOR_CONTROL_PASSWORD must be set in .env}
|
|
CRAWLER_TOR_RECIRCUIT_MAX_ATTEMPTS: ${CRAWLER_TOR_RECIRCUIT_MAX_ATTEMPTS:-3}
|
|
# Tor cookie-file auth alternative — the controller prefers cookie
|
|
# when both are present. Leave unset for the bundled
|
|
# HashedControlPassword path above.
|
|
CRAWLER_TOR_CONTROL_COOKIE_PATH: ${CRAWLER_TOR_CONTROL_COOKIE_PATH:-}
|
|
# Analysis worker — boot seeds + env-only API key.
|
|
ANALYSIS_ENABLED: ${ANALYSIS_ENABLED:-false}
|
|
# Engine selection (deploy-time): `ocr` (in-process ocrs, default).
|
|
# The `vision` (local LLM) backend is temporarily disabled — the worker
|
|
# forces OCR regardless, so a `vision` override here is currently inert.
|
|
# Model paths default to the image-baked /models.
|
|
ANALYSIS_BACKEND: ${ANALYSIS_BACKEND:-ocr}
|
|
OCRS_DETECTION_MODEL: ${OCRS_DETECTION_MODEL:-/models/text-detection.rten}
|
|
OCRS_RECOGNITION_MODEL: ${OCRS_RECOGNITION_MODEL:-/models/text-recognition.rten}
|
|
ANALYSIS_VISION_URL: ${ANALYSIS_VISION_URL:-}
|
|
ANALYSIS_VISION_MODEL: ${ANALYSIS_VISION_MODEL:-}
|
|
ANALYSIS_WORKERS: ${ANALYSIS_WORKERS:-1}
|
|
ANALYSIS_JOB_TIMEOUT_SECS: ${ANALYSIS_JOB_TIMEOUT_SECS:-600}
|
|
# API key never gets persisted to app_settings; bearer-attached to
|
|
# every vision call.
|
|
ANALYSIS_API_KEY: ${ANALYSIS_API_KEY:-}
|
|
# Analysis tuning knobs — boot seeds, then live-editable in dashboard.
|
|
ANALYSIS_MAX_TOKENS: ${ANALYSIS_MAX_TOKENS:-4096}
|
|
ANALYSIS_MAX_PIXELS: ${ANALYSIS_MAX_PIXELS:-1000000}
|
|
ANALYSIS_MIN_SLICE_HEIGHT: ${ANALYSIS_MIN_SLICE_HEIGHT:-640}
|
|
ANALYSIS_SLICE_OVERLAP: ${ANALYSIS_SLICE_OVERLAP:-0.12}
|
|
ANALYSIS_TALL_ASPECT: ${ANALYSIS_TALL_ASPECT:-1.6}
|
|
ANALYSIS_MAX_SLICES: ${ANALYSIS_MAX_SLICES:-16}
|
|
ANALYSIS_MAX_IMAGE_BYTES: ${ANALYSIS_MAX_IMAGE_BYTES:-8388608}
|
|
ANALYSIS_OCR_MAX_DECODE_PIXELS: ${ANALYSIS_OCR_MAX_DECODE_PIXELS:-100000000}
|
|
ANALYSIS_RESPONSE_FORMAT: ${ANALYSIS_RESPONSE_FORMAT:-json_schema}
|
|
ANALYSIS_FREQUENCY_PENALTY: ${ANALYSIS_FREQUENCY_PENALTY:-0.3}
|
|
ANALYSIS_TEMPERATURE: ${ANALYSIS_TEMPERATURE:-0.0}
|
|
# Vision readiness gate (env-ONLY, not a dashboard setting). When set,
|
|
# the analysis worker refuses to lease a page until this answers 2xx, so
|
|
# the vision-manager autoscaler can idle-stop the vision container
|
|
# without jobs burning their retries / landing `failed` rows. Leave
|
|
# empty (the default) to disable the gate for an always-on endpoint.
|
|
# Pair with the `ai` profile services below. See VISION-AUTOSCALE.md.
|
|
ANALYSIS_VISION_HEALTH_URL: ${ANALYSIS_VISION_HEALTH_URL:-}
|
|
volumes:
|
|
- storage-data:/var/lib/mangalord/storage
|
|
# No host port mapping in the default setup — the frontend proxies
|
|
# /api/* through its hooks.server.ts. Expose :8080 only if you want
|
|
# to hit the API directly from the host (e.g., bot scripts during
|
|
# development).
|
|
expose:
|
|
- "8080"
|
|
restart: unless-stopped
|
|
|
|
frontend:
|
|
build: ./frontend
|
|
depends_on:
|
|
- backend
|
|
environment:
|
|
# SvelteKit's hooks.server.ts proxies /api/* to this URL so the
|
|
# browser only ever talks to :3000 and cookies stay same-origin.
|
|
# `.env` override exists for deploys where the SvelteKit container
|
|
# needs to reach a backend on a different hostname (typical with
|
|
# an external reverse proxy in front of axum).
|
|
BACKEND_URL: ${BACKEND_URL:-http://backend:8080}
|
|
# Per-request wall-clock cap for the /api/* reverse proxy. Defaults
|
|
# to 300000 (5 min) in hooks.server.ts; raise/lower via .env.
|
|
BACKEND_PROXY_TIMEOUT_MS: ${BACKEND_PROXY_TIMEOUT_MS:-300000}
|
|
ports:
|
|
- "3000:3000"
|
|
restart: unless-stopped
|
|
|
|
# ----- Vision autoscaling (profile: ai) -----------------------------------
|
|
# Two extra containers that idle-stop the mangalord-vision (llama.cpp)
|
|
# container when there is no analysis work and start it back up on demand.
|
|
# Gated behind `profiles: [ai]` so a vanilla `docker compose up` is
|
|
# unaffected — bring them up with `docker compose --profile ai up -d`.
|
|
# The vision container itself is NOT defined here (it lives elsewhere on the
|
|
# host); the manager drives it by name. See VISION-AUTOSCALE.md.
|
|
|
|
# Scoped Docker access for the manager. The proxy gates by API *section*
|
|
# (not per-method), so CONTAINERS=1 + POST=1 permits the full /containers
|
|
# lifecycle — inspect/start/stop, but also create/kill/restart/update/
|
|
# rename/remove. It does NOT expose exec, images, volumes, networks, swarm,
|
|
# etc. (all default-denied). The trust boundary is therefore: (a) this is an
|
|
# internal-only network reachable solely by vision-manager, and (b) the raw
|
|
# host socket is mounted HERE and nowhere else — never on the backend. A
|
|
# backend RCE still cannot reach the Docker API. If you need true start/stop-
|
|
# only granularity, front the socket with an allow-list reverse proxy instead.
|
|
docker-socket-proxy:
|
|
image: tecnativa/docker-socket-proxy:latest
|
|
profiles: ["ai"]
|
|
environment:
|
|
CONTAINERS: 1 # allow the /containers/* section
|
|
POST: 1 # allow write methods (start/stop are POSTs)
|
|
# Everything else stays at its default-deny (EXEC, IMAGES, NETWORKS, ...).
|
|
volumes:
|
|
- /var/run/docker.sock:/var/run/docker.sock:ro
|
|
networks:
|
|
- vision-internal
|
|
restart: unless-stopped
|
|
|
|
vision-manager:
|
|
build: ./vision-manager
|
|
profiles: ["ai"]
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
docker-socket-proxy:
|
|
condition: service_started
|
|
environment:
|
|
# Read-only role — apply vision-manager/readonly-role.sql once, then set
|
|
# VISION_MANAGER_DATABASE_URL in .env. Do NOT reuse the backend creds.
|
|
DATABASE_URL: ${VISION_MANAGER_DATABASE_URL:?set VISION_MANAGER_DATABASE_URL in .env (vision_manager read-only role)}
|
|
VISION_CONTAINER: ${VISION_CONTAINER:-mangalord-vision}
|
|
VISION_HEALTH_URL: ${VISION_HEALTH_URL:-http://mangalord-vision:8000/health}
|
|
DOCKER_HOST: tcp://docker-socket-proxy:2375
|
|
POLL_INTERVAL: ${VISION_POLL_INTERVAL:-20}
|
|
STOP_DEBOUNCE: ${VISION_STOP_DEBOUNCE:-600}
|
|
START_HEALTH_TIMEOUT: ${VISION_START_HEALTH_TIMEOUT:-300}
|
|
RESPECT_CRAWL_MUTEX: ${VISION_RESPECT_CRAWL_MUTEX:-1}
|
|
MAX_UPTIME: ${VISION_MAX_UPTIME:-0}
|
|
# Memory-pressure yield — stop vision when the HOST is short on RAM so a
|
|
# spike elsewhere can finish without the kernel OOM-killer. See
|
|
# VISION-MEMORY-YIELD.md.
|
|
MEM_YIELD_ENABLED: ${VISION_MEM_YIELD_ENABLED:-1}
|
|
MEM_HIGH_WATERMARK_PCT: ${VISION_MEM_HIGH_WATERMARK_PCT:-92}
|
|
MEM_LOW_WATERMARK_PCT: ${VISION_MEM_LOW_WATERMARK_PCT:-80}
|
|
MEM_YIELD_COOLDOWN: ${VISION_MEM_YIELD_COOLDOWN:-300}
|
|
MEM_POLL_INTERVAL: ${VISION_MEM_POLL_INTERVAL:-5}
|
|
networks:
|
|
- default # reach postgres + the vision container by name
|
|
- vision-internal # reach the socket-proxy
|
|
restart: unless-stopped
|
|
|
|
networks:
|
|
default:
|
|
# Internal-only: no route to the outside world. Only the manager and the
|
|
# socket-proxy sit on it, so nothing else can reach the Docker API.
|
|
vision-internal:
|
|
internal: true
|
|
|
|
volumes:
|
|
postgres-data:
|
|
storage-data:
|