Compare commits
148 Commits
1685bf105c
...
fix/deploy
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9b90929269 | ||
|
|
117e67fa80 | ||
|
|
f275de5c8f | ||
|
|
3f0f9c098b | ||
|
|
e52b2f1cd1 | ||
|
|
5969ec74ea | ||
|
|
a4bac03628 | ||
|
|
6fd75adb27 | ||
|
|
7d0334bf22 | ||
|
|
402215d405 | ||
|
|
2b313e67e0 | ||
|
|
2551c25436 | ||
|
|
d51c6b8c4b | ||
|
|
3bcb7c6a76 | ||
|
|
08d92b7531 | ||
|
|
06bc9ddcb3 | ||
|
|
5f702f2b40 | ||
|
|
31faccfdf8 | ||
|
|
06ade4e158 | ||
|
|
b601c062bd | ||
|
|
0c0eed885a | ||
|
|
a77c2ddc00 | ||
|
|
40c6fd2ccb | ||
|
|
e69ec4d736 | ||
|
|
3fb1b5d80d | ||
|
|
e1ca9d192f | ||
|
|
5009590882 | ||
|
|
d9738a4cb9 | ||
|
|
669a191968 | ||
|
|
a1733b03d5 | ||
|
|
4026648f98 | ||
|
|
44641473ea | ||
|
|
57a907eca5 | ||
|
|
6e0a760271 | ||
|
|
0abf413693 | ||
|
|
002355ba40 | ||
|
|
6155b4123d | ||
|
|
7758270cac | ||
|
|
9b8698f86b | ||
|
|
3c3a7d0082 | ||
|
|
3654aca18b | ||
|
|
f243bfe89a | ||
|
|
5546fb82e6 | ||
|
|
e2b7e54af9 | ||
|
|
15d338eeb8 | ||
|
|
461b1eaf65 | ||
|
|
460258c451 | ||
|
|
bbdfae09a0 | ||
|
|
f8cba95e49 | ||
|
|
4d14df18d0 | ||
|
|
ee554e7f38 | ||
|
|
0fa40ddf80 | ||
|
|
c48d43f5b3 | ||
|
|
db7c4459d7 | ||
|
|
dd7b05415e | ||
|
|
d2ad560df2 | ||
|
|
d452afb00e | ||
|
|
3f74c65787 | ||
|
|
02971f3186 | ||
|
|
e5201a9889 | ||
|
|
c229b560d8 | ||
|
|
af997a84dd | ||
|
|
2bef6e19ef | ||
|
|
db88230221 | ||
|
|
bf68bc08f4 | ||
|
|
38e34fddf1 | ||
|
|
3c683247c0 | ||
|
|
0447a6ad0e | ||
|
|
9666d74a46 | ||
|
|
5affef47cf | ||
|
|
8e906d7866 | ||
|
|
1148c2e906 | ||
|
|
9f3712894d | ||
|
|
c197b2c025 | ||
|
|
d643256f36 | ||
|
|
99f79e2898 | ||
|
|
df275bbefa | ||
|
|
768e712a26 | ||
|
|
811c724685 | ||
|
|
c647ddfa6b | ||
|
|
36fe59caa5 | ||
|
|
641174717c | ||
|
|
16d1f356be | ||
|
|
a4b2c5bd1c | ||
|
|
683b1b6f65 | ||
|
|
d6c91974eb | ||
|
|
4cdb3ae14a | ||
|
|
faf7a2504a | ||
|
|
14c667c694 | ||
|
|
3f6dafba05 | ||
|
|
0ed97f45cf | ||
|
|
b3d876fa85 | ||
|
|
80179357a7 | ||
|
|
dba4d3f932 | ||
|
|
239459bb91 | ||
|
|
41ac742af8 | ||
|
|
a084ba86c5 | ||
|
|
91ff961850 | ||
|
|
f08d858281 | ||
|
|
498b4e256b | ||
|
|
5d21b3eefd | ||
|
|
74849c8d50 | ||
|
|
6ed0aaf0d7 | ||
|
|
226e455328 | ||
|
|
2c44006392 | ||
|
|
b1e2e66305 | ||
|
|
ec64fc361b | ||
|
|
5f523d55fe | ||
|
|
bbdb45155e | ||
|
|
564104ae23 | ||
|
|
104c3dde16 | ||
|
|
723a492d44 | ||
|
|
fabc6af656 | ||
|
|
136417d6b4 | ||
|
|
d4aa7b4932 | ||
|
|
8f6e1d4ff7 | ||
|
|
da586d0978 | ||
|
|
20125fb713 | ||
|
|
6ccf2d0011 | ||
|
|
a3c0082c4b | ||
|
|
792a4f0e4b | ||
|
|
b32231d4f4 | ||
|
|
1c7495e568 | ||
|
|
d9025f783a | ||
|
|
a490642f5f | ||
|
|
8f5afb9df7 | ||
|
|
0d7938aff5 | ||
|
|
e79e020566 | ||
|
|
1bb58b59ad | ||
|
|
edcef0258c | ||
|
|
bbec815854 | ||
|
|
309c25bc06 | ||
|
|
b241ba6415 | ||
|
|
d228676a56 | ||
|
|
2340f21637 | ||
|
|
e42d8a92a1 | ||
|
|
1cdab21514 | ||
|
|
05f76514a2 | ||
|
|
e6efffafe5 | ||
|
|
16d8bdb680 | ||
|
|
2761ac7db6 | ||
|
|
e619a3bd64 | ||
|
|
8a769b52bf | ||
|
|
251f9f1469 | ||
|
|
2e98f5ddf5 | ||
|
|
141c918dd5 | ||
|
|
9a0ceeced7 | ||
|
|
f7fdfa4627 |
9
.claude/settings.json
Normal file
9
.claude/settings.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"permissions": {
|
||||
"allow": [
|
||||
"Bash(cargo check *)",
|
||||
"Bash(cargo clippy *)",
|
||||
"Bash(git --no-pager diff *)"
|
||||
]
|
||||
}
|
||||
}
|
||||
95
.env.example
95
.env.example
@@ -4,12 +4,40 @@ DOMAIN=my-event.example.com
|
||||
|
||||
# ── App server ────────────────────────────────────────────────────────────────
|
||||
APP_PORT=3000
|
||||
# Set to `production` in real deployments. This activates the secret guard that
|
||||
# refuses to boot with placeholder JWT_SECRET / ADMIN_PASSWORD_HASH values.
|
||||
# (docker-compose.yml already sets APP_ENV=production for the app service.)
|
||||
APP_ENV=production
|
||||
|
||||
# ── Database ──────────────────────────────────────────────────────────────────
|
||||
DATABASE_URL=postgres://eventsnap:secret@db:5432/eventsnap
|
||||
# Set a strong password and keep it in sync between DATABASE_URL and
|
||||
# POSTGRES_PASSWORD. Generate one with: openssl rand -hex 24
|
||||
#
|
||||
# SET THIS BEFORE THE FIRST `docker compose up -d`. Postgres reads POSTGRES_PASSWORD
|
||||
# only when it initialises its data directory, on that very first boot. Change it
|
||||
# afterwards and the app authenticates with the new password against a volume still
|
||||
# holding the old one — a permanent restart loop ("password authentication failed").
|
||||
# The only ways out are restoring the old password or `docker compose down -v`, which
|
||||
# deletes the database, the media and the exports. In production the app refuses to
|
||||
# boot while this is still the placeholder below, so it cannot be missed by accident.
|
||||
DATABASE_URL=postgres://eventsnap:CHANGE_ME_use_a_strong_password@db:5432/eventsnap
|
||||
POSTGRES_USER=eventsnap
|
||||
POSTGRES_PASSWORD=secret
|
||||
POSTGRES_PASSWORD=CHANGE_ME_use_a_strong_password
|
||||
POSTGRES_DB=eventsnap
|
||||
# Connection pool size. Default 10. For a busy event (~100 guests polling the feed
|
||||
# + SSE + uploads at once) raise to ~30 so requests don't queue on a pool permit.
|
||||
# PAIRED WITH THE DB CONTAINER'S MEMORY LIMIT: 30 backends plus Postgres 16's default
|
||||
# shared_buffers is already snug in the 1G that docker-compose.yml allots the `db`
|
||||
# service. If you raise this, raise `db.deploy.resources.limits.memory` with it — an
|
||||
# OOM in Postgres doesn't degrade one feature, it takes the whole event down.
|
||||
DATABASE_MAX_CONNECTIONS=30
|
||||
|
||||
# Log level. `info` is the right production default: at `debug` the tower-http trace
|
||||
# layer writes a line per request AND per response, which on a busy event is a large
|
||||
# multiple of the useful output. Container logs are capped at 10m x 3 per service
|
||||
# (docker-compose.yml), so a chatty level buys you a shorter history, not more of it.
|
||||
# To debug a live event: RUST_LOG=eventsnap_backend=debug docker compose up -d app
|
||||
RUST_LOG=info
|
||||
|
||||
# ── Authentication ────────────────────────────────────────────────────────────
|
||||
# Generate with: openssl rand -hex 64
|
||||
@@ -17,8 +45,14 @@ JWT_SECRET=change_me_to_a_random_64_byte_hex_string
|
||||
SESSION_EXPIRY_DAYS=30
|
||||
|
||||
# Admin dashboard password (bcrypt hash).
|
||||
# Generate with: htpasswd -bnBC 12 "" yourpassword | tr -d ':\n'
|
||||
ADMIN_PASSWORD_HASH=$2y$12$placeholder_replace_me
|
||||
# Generate with an image the stack already pulls (htpasswd needs apache2-utils, which
|
||||
# a stock VPS does not have):
|
||||
# docker run --rm caddy:2-alpine caddy hash-password --plaintext 'yourpassword'
|
||||
# IMPORTANT: keep the SINGLE QUOTES. A bcrypt hash is full of `$` (e.g. $2b$12$…$…),
|
||||
# and both Docker Compose's env_file interpolation and dotenvy's variable substitution
|
||||
# would otherwise eat the `$…` segments (reading them as unset vars) and corrupt the
|
||||
# hash — every admin login then 401s. Single quotes make both read it literally.
|
||||
ADMIN_PASSWORD_HASH='$2y$12$placeholder_replace_me'
|
||||
|
||||
# ── Event ─────────────────────────────────────────────────────────────────────
|
||||
EVENT_NAME=Max & Maria's Wedding
|
||||
@@ -26,20 +60,47 @@ EVENT_SLUG=max-maria-2026
|
||||
|
||||
# ── Storage ───────────────────────────────────────────────────────────────────
|
||||
MEDIA_PATH=/media
|
||||
# Export archives (Gallery.zip / Memories.zip). MUST be outside MEDIA_PATH —
|
||||
# /media is publicly served, so exports here would be downloadable without auth.
|
||||
EXPORT_PATH=/exports
|
||||
|
||||
# ── Upload limits ─────────────────────────────────────────────────────────────
|
||||
DEFAULT_MAX_IMAGE_SIZE_MB=20
|
||||
DEFAULT_MAX_VIDEO_SIZE_MB=500
|
||||
|
||||
# ── Rate limiting ─────────────────────────────────────────────────────────────
|
||||
DEFAULT_UPLOAD_RATE_PER_HOUR=10
|
||||
DEFAULT_FEED_RATE_PER_MIN=60
|
||||
DEFAULT_EXPORT_RATE_PER_DAY=3
|
||||
|
||||
# ── Capacity ──────────────────────────────────────────────────────────────────
|
||||
DEFAULT_ESTIMATED_GUEST_COUNT=100
|
||||
# Fraction of total storage that triggers the "low storage" warning (0.0–1.0)
|
||||
DEFAULT_QUOTA_TOLERANCE=0.75
|
||||
# ── Runtime settings (upload limits, rate limits, capacity) ───────────────────
|
||||
# NOTE: These are NOT environment variables. Upload size caps, rate limits, guest
|
||||
# count and quota tolerance are stored in the database `config` table (seeded once
|
||||
# at first boot) and changed at runtime from the ADMIN DASHBOARD — the backend does
|
||||
# not read them from .env. Setting them here has no effect. Current seeded defaults:
|
||||
# upload rate 100 / hour / guest (raised from 10 by migration 015)
|
||||
# feed rate 60 / minute
|
||||
# export rate 3 / day
|
||||
# max image size 20 MB
|
||||
# max video size 500 MB
|
||||
# estimated guests 100
|
||||
# quota tolerance 0.75 (see below — NOT a warning threshold)
|
||||
# Adjust these in the admin UI before the event if needed.
|
||||
#
|
||||
# quota_tolerance is the MULTIPLIER IN THE PER-USER QUOTA FORMULA, not the point at
|
||||
# which anything warns you:
|
||||
#
|
||||
# per_user_limit = floor(free_disk * quota_tolerance / active_uploaders)
|
||||
#
|
||||
# It is recomputed against LIVE free space on every upload, so it self-throttles: guests
|
||||
# converge on a fixed point at tolerance/(1+tolerance) of the free space you started
|
||||
# with — 43% at 0.75, i.e. ~30 GB of a fresh 70 GB.
|
||||
#
|
||||
# Raising it therefore AUTHORISES GUESTS TO FILL MORE OF THE DISK. Setting 0.95 in the
|
||||
# belief that it means "warn me later" moves the fixed point to ~49% and eats the
|
||||
# headroom the keepsake needs — and the keepsake needs a lot, because Gallery.zip and
|
||||
# Memories.zip are each roughly a second copy of every original (both store media
|
||||
# uncompressed). Budget for media + 2x media, or move exports to their own volume.
|
||||
#
|
||||
# 0.75 is the tested default. Lower it if the box is tight; raise it only if you have
|
||||
# provisioned export headroom separately.
|
||||
|
||||
# ── Workers ───────────────────────────────────────────────────────────────────
|
||||
# Number of parallel image/video compression workers. Default 2. This is the main
|
||||
# throughput bottleneck: with 2 workers a burst of uploads can take ~5s to appear.
|
||||
# For a large event (100+ guests) 4 is a good target — but each worker can run an
|
||||
# ffmpeg transcode, so if you raise this ALSO raise the app container's memory limit
|
||||
# in docker-compose.yml (`app.deploy.resources.limits.memory`) from 1G to ~2G, or a
|
||||
# burst of large videos can OOM the box and take Postgres down with it.
|
||||
COMPRESSION_WORKER_CONCURRENCY=2
|
||||
|
||||
45
.env.test
45
.env.test
@@ -1,45 +0,0 @@
|
||||
# ── Domain ────────────────────────────────────────────────────────────────────
|
||||
# Public domain Caddy will serve and obtain a TLS certificate for.
|
||||
DOMAIN=my-event.example.com
|
||||
|
||||
# ── App server ────────────────────────────────────────────────────────────────
|
||||
APP_PORT=3000
|
||||
|
||||
# ── Database ──────────────────────────────────────────────────────────────────
|
||||
DATABASE_URL=postgres://eventsnap:secret@db:5432/eventsnap
|
||||
POSTGRES_USER=eventsnap
|
||||
POSTGRES_PASSWORD=secret
|
||||
POSTGRES_DB=eventsnap
|
||||
|
||||
# ── Authentication ────────────────────────────────────────────────────────────
|
||||
# Generate with: openssl rand -hex 64
|
||||
JWT_SECRET=change_me_to_a_random_64_byte_hex_string
|
||||
SESSION_EXPIRY_DAYS=30
|
||||
|
||||
# Admin dashboard password (bcrypt hash).
|
||||
# Generate with: htpasswd -bnBC 12 "" yourpassword | tr -d ':\n'
|
||||
ADMIN_PASSWORD_HASH=$2y$12$placeholder_replace_me
|
||||
|
||||
# ── Event ─────────────────────────────────────────────────────────────────────
|
||||
EVENT_NAME=Max & Maria's Wedding
|
||||
EVENT_SLUG=max-maria-2026
|
||||
|
||||
# ── Storage ───────────────────────────────────────────────────────────────────
|
||||
MEDIA_PATH=/media
|
||||
|
||||
# ── Upload limits ─────────────────────────────────────────────────────────────
|
||||
DEFAULT_MAX_IMAGE_SIZE_MB=20
|
||||
DEFAULT_MAX_VIDEO_SIZE_MB=500
|
||||
|
||||
# ── Rate limiting ─────────────────────────────────────────────────────────────
|
||||
DEFAULT_UPLOAD_RATE_PER_HOUR=10
|
||||
DEFAULT_FEED_RATE_PER_MIN=60
|
||||
DEFAULT_EXPORT_RATE_PER_DAY=3
|
||||
|
||||
# ── Capacity ──────────────────────────────────────────────────────────────────
|
||||
DEFAULT_ESTIMATED_GUEST_COUNT=100
|
||||
# Fraction of total storage that triggers the "low storage" warning (0.0–1.0)
|
||||
DEFAULT_QUOTA_TOLERANCE=0.75
|
||||
|
||||
# ── Workers ───────────────────────────────────────────────────────────────────
|
||||
COMPRESSION_WORKER_CONCURRENCY=2
|
||||
73
.github/workflows/audit.yml
vendored
Normal file
73
.github/workflows/audit.yml
vendored
Normal file
@@ -0,0 +1,73 @@
|
||||
# Dependency advisory scan.
|
||||
#
|
||||
# Lives in .github/workflows/ because Gitea Actions scans BOTH .gitea/workflows and
|
||||
# .github/workflows, and e2e.yml already proves this instance resolves `actions/*` from GitHub
|
||||
# (DEFAULT_ACTIONS_URL). Keeping one directory avoids a split where half the CI is invisible
|
||||
# depending on which convention you look under.
|
||||
#
|
||||
# Unlike the GitHub-hosted runners this workflow does NOT assume a preinstalled Rust toolchain —
|
||||
# the common Gitea runner images (gitea/runner-images, catthehacker/ubuntu) ship Node and Docker
|
||||
# but not cargo. So we install it explicitly instead of relying on the ambient environment.
|
||||
name: Audit
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches: [main]
|
||||
schedule:
|
||||
# New advisories land against unchanged code, so a push-only trigger would never see them.
|
||||
- cron: '0 6 * * 1'
|
||||
|
||||
jobs:
|
||||
cargo-audit:
|
||||
name: cargo audit (backend)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust toolchain
|
||||
run: |
|
||||
if ! command -v cargo > /dev/null; then
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --profile minimal
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
fi
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/bin
|
||||
key: cargo-audit-${{ hashFiles('backend/Cargo.lock') }}
|
||||
restore-keys: cargo-audit-
|
||||
|
||||
- name: Install cargo-audit
|
||||
run: cargo install cargo-audit --locked || true
|
||||
|
||||
# `rsa` 0.9 (RUSTSEC-2023-0071, "Marvin") is a transitive dep of the SQLx MySQL driver that
|
||||
# this app never exercises: auth is HS256 + bcrypt, and the only database is Postgres. There
|
||||
# is no patched release, so failing the build on it would mean a permanently red pipeline
|
||||
# that everyone learns to ignore — which is worse than not scanning at all. Ignore it
|
||||
# SPECIFICALLY, so that any OTHER advisory still fails the job.
|
||||
- name: Audit
|
||||
working-directory: ./backend
|
||||
run: cargo audit --deny warnings --ignore RUSTSEC-2023-0071
|
||||
|
||||
npm-audit:
|
||||
name: npm audit (frontend)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Install deps
|
||||
working-directory: ./frontend
|
||||
run: npm install
|
||||
# Production dependencies only: a dev-only advisory (build tooling, test runners) can't be
|
||||
# reached by a guest at the party, and gating merges on it just trains people to skip the gate.
|
||||
- name: Audit
|
||||
working-directory: ./frontend
|
||||
run: npm audit --omit=dev --audit-level=high
|
||||
129
.github/workflows/checks.yml
vendored
Normal file
129
.github/workflows/checks.yml
vendored
Normal file
@@ -0,0 +1,129 @@
|
||||
# The checks that were only ever running on a laptop.
|
||||
#
|
||||
# Before this file, CI ran Playwright (chromium-desktop) and the dependency audit — and nothing
|
||||
# else. `cargo test` (40 tests), the frontend vitest suite (5 files, including the offline
|
||||
# upload-queue and auth-token logic), svelte-check, and the e2e typecheck were all green on a
|
||||
# developer's machine and gated NOTHING. A check that only ever runs locally is not running.
|
||||
#
|
||||
# In .github/workflows/ because Gitea Actions scans it too (see audit.yml). Rust is installed
|
||||
# explicitly: the common Gitea runner images ship Node and Docker, not cargo.
|
||||
name: Checks
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
backend:
|
||||
name: Backend — cargo test + clippy + fmt
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust toolchain
|
||||
run: |
|
||||
if ! command -v cargo > /dev/null; then
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --profile minimal
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
fi
|
||||
rustup component add clippy rustfmt
|
||||
|
||||
- uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
backend/target
|
||||
key: cargo-${{ hashFiles('backend/Cargo.lock') }}
|
||||
restore-keys: cargo-
|
||||
|
||||
# SQLx runs its queries against a live database at TEST time (see backend/tests/), so the
|
||||
# DB-backed tests need one. The pure unit tests don't care, but starting it unconditionally
|
||||
# keeps the job simple and honest about what it covers.
|
||||
- name: Start Postgres
|
||||
run: |
|
||||
docker run -d --name ci-pg -p 5432:5432 \
|
||||
-e POSTGRES_PASSWORD=postgres -e POSTGRES_DB=eventsnap_ci \
|
||||
postgres:16-alpine
|
||||
for _ in $(seq 1 30); do
|
||||
docker exec ci-pg pg_isready -U postgres > /dev/null 2>&1 && break
|
||||
sleep 1
|
||||
done
|
||||
|
||||
- name: Test
|
||||
working-directory: ./backend
|
||||
env:
|
||||
DATABASE_URL: postgres://postgres:postgres@localhost:5432/eventsnap_ci
|
||||
# `#[sqlx::test]` creates a fresh database per test and opens its own pool; run at unbounded
|
||||
# parallelism against a default `max_connections=100` Postgres, a full suite can exhaust the
|
||||
# server's connection slots and fail with PoolTimedOut — a pure infra flake, not a real
|
||||
# failure. Cap the concurrency so the gate stays trustworthy on a loaded runner.
|
||||
run: cargo test --all-features -- --test-threads=8
|
||||
|
||||
- name: Clippy
|
||||
working-directory: ./backend
|
||||
run: cargo clippy --all-targets -- -D warnings
|
||||
|
||||
- name: Format
|
||||
working-directory: ./backend
|
||||
run: cargo fmt --check
|
||||
|
||||
frontend:
|
||||
name: Frontend — vitest + svelte-check
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: 'npm'
|
||||
cache-dependency-path: 'frontend/package-lock.json'
|
||||
|
||||
- name: Install deps
|
||||
working-directory: ./frontend
|
||||
run: npm ci || npm install
|
||||
|
||||
- name: Unit tests
|
||||
working-directory: ./frontend
|
||||
run: npm run test:unit
|
||||
|
||||
- name: svelte-check
|
||||
working-directory: ./frontend
|
||||
run: npx svelte-check --threshold error
|
||||
|
||||
- name: ESLint
|
||||
working-directory: ./frontend
|
||||
run: npm run lint
|
||||
|
||||
- name: Prettier
|
||||
working-directory: ./frontend
|
||||
run: npm run format:check
|
||||
|
||||
e2e-typecheck:
|
||||
name: E2E — typecheck + lint
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Install deps
|
||||
working-directory: ./e2e
|
||||
run: npm install
|
||||
# Playwright TRANSPILES specs without typechecking them, so a type error in a spec is
|
||||
# invisible until the assertion it guards silently does the wrong thing at runtime.
|
||||
- name: tsc --noEmit
|
||||
working-directory: ./e2e
|
||||
run: npx tsc --noEmit
|
||||
|
||||
- name: ESLint
|
||||
working-directory: ./e2e
|
||||
run: npm run lint
|
||||
|
||||
- name: Prettier
|
||||
working-directory: ./e2e
|
||||
run: npm run format:check
|
||||
134
.github/workflows/e2e.yml
vendored
Normal file
134
.github/workflows/e2e.yml
vendored
Normal file
@@ -0,0 +1,134 @@
|
||||
name: E2E
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
e2e:
|
||||
name: Playwright E2E (chromium + webkit)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: 'npm'
|
||||
cache-dependency-path: 'e2e/package.json'
|
||||
|
||||
- name: Install e2e deps
|
||||
working-directory: ./e2e
|
||||
run: npm install
|
||||
|
||||
- name: Install Playwright browsers
|
||||
working-directory: ./e2e
|
||||
run: npx playwright install --with-deps chromium webkit
|
||||
|
||||
- name: Bring up the test stack
|
||||
working-directory: ./e2e
|
||||
run: docker compose -f docker-compose.test.yml up -d --build
|
||||
|
||||
- name: Wait for stack health
|
||||
working-directory: ./e2e
|
||||
run: |
|
||||
for i in $(seq 1 60); do
|
||||
if curl -fsS http://localhost:3101/health > /dev/null; then exit 0; fi
|
||||
sleep 2
|
||||
done
|
||||
echo "Stack never became healthy. Dumping logs:"
|
||||
docker compose -f docker-compose.test.yml logs
|
||||
exit 1
|
||||
|
||||
- name: Run E2E tests
|
||||
working-directory: ./e2e
|
||||
run: npm run test:e2e -- --project=chromium-desktop
|
||||
|
||||
# 09-mobile is `testIgnore`d on chromium-desktop (it needs hasTouch + a phone viewport), so
|
||||
# running only that project left 22 tests — focus traps, touch targets, safe-area insets,
|
||||
# viewport reflow, upload-cancel — never executing in CI. On a phone-first event app, where
|
||||
# essentially every real guest is on a phone, that was the wrong half of the suite to skip.
|
||||
- name: Run E2E tests (mobile)
|
||||
working-directory: ./e2e
|
||||
run: npm run test:e2e -- --project=chromium-mobile
|
||||
|
||||
# iOS Safari is the app's stated primary user (a wedding guest opening a QR link), and
|
||||
# WebKit is the ONLY engine here that reproduces two of its behaviours:
|
||||
# - it enforces X-Frame-Options on the download iframe, so a site-wide `DENY` makes the
|
||||
# keepsake download silently do nothing. Blink hands attachments to the download
|
||||
# manager first and never notices. That shipped once already.
|
||||
# - it abandons a <video> load without a 206 response to its Range probe.
|
||||
# Both regressions are invisible to every Chromium project, so running WebKit is what
|
||||
# actually gates them on a PR rather than on someone remembering to test locally.
|
||||
#
|
||||
# The project is scoped in playwright.config.ts to the journeys a guest walks
|
||||
# (01-auth, 02-upload, 03-feed, 06-export); four IndexedDB-blob tests skip themselves
|
||||
# there — see helpers/webkit.ts for why that is the harness and not the app.
|
||||
- name: Run E2E tests (webkit / iOS)
|
||||
working-directory: ./e2e
|
||||
run: npm run test:e2e -- --project=webkit-iphone
|
||||
|
||||
- name: Upload Playwright report
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: playwright-report
|
||||
path: e2e/playwright-report/
|
||||
retention-days: 14
|
||||
|
||||
- name: Upload traces
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: traces
|
||||
path: e2e/test-results/
|
||||
retention-days: 14
|
||||
|
||||
- name: Tear down stack
|
||||
if: always()
|
||||
working-directory: ./e2e
|
||||
run: docker compose -f docker-compose.test.yml down -v
|
||||
|
||||
smoke-ua-matrix:
|
||||
name: Cross-UA smoke matrix
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
- name: Install e2e deps
|
||||
working-directory: ./e2e
|
||||
run: npm install
|
||||
- name: Install Playwright browsers (full)
|
||||
working-directory: ./e2e
|
||||
run: npx playwright install --with-deps chromium firefox webkit
|
||||
- name: Bring up the test stack
|
||||
working-directory: ./e2e
|
||||
run: docker compose -f docker-compose.test.yml up -d --build
|
||||
- name: Wait for stack health
|
||||
working-directory: ./e2e
|
||||
run: |
|
||||
for i in $(seq 1 60); do
|
||||
if curl -fsS http://localhost:3101/health > /dev/null; then exit 0; fi
|
||||
sleep 2
|
||||
done
|
||||
docker compose -f docker-compose.test.yml logs
|
||||
exit 1
|
||||
- name: Smoke matrix
|
||||
working-directory: ./e2e
|
||||
run: npm run test:e2e:smoke
|
||||
- name: Upload smoke report
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: playwright-smoke-report
|
||||
path: e2e/playwright-report/
|
||||
retention-days: 14
|
||||
- name: Tear down
|
||||
if: always()
|
||||
working-directory: ./e2e
|
||||
run: docker compose -f docker-compose.test.yml down -v
|
||||
27
.gitignore
vendored
27
.gitignore
vendored
@@ -1,5 +1,7 @@
|
||||
# Environment secrets — never commit the real .env
|
||||
.env
|
||||
# Stale local scratch copy of .env.example; nothing in the test stack reads it.
|
||||
.env.test
|
||||
|
||||
# Rust
|
||||
backend/target/
|
||||
@@ -11,9 +13,30 @@ frontend/build/
|
||||
frontend/export-viewer/node_modules/
|
||||
frontend/export-viewer/.svelte-kit/
|
||||
|
||||
# Media uploads (mounted volume in production)
|
||||
media/
|
||||
# Media uploads. In production these live in the `media_data` DOCKER VOLUME, never in the
|
||||
# working tree — so this pattern is anchored to the repo root and exists only for a local
|
||||
# bind-mount experiment.
|
||||
#
|
||||
# It used to read `media/`, unanchored, which matches a directory of that name at ANY depth.
|
||||
# The only one in the repo is `e2e/fixtures/media/`, so the rule's entire practical effect was
|
||||
# to keep every E2E fixture untracked: a fresh clone got the specs and none of the images or
|
||||
# videos they read. `.github/workflows/e2e.yml` does a plain checkout and generates nothing, so
|
||||
# the committed CI job could not have run the upload, video or export suites at all.
|
||||
/media/
|
||||
|
||||
# Playwright E2E suite — runtime artifacts (the suite itself is committed)
|
||||
e2e/node_modules/
|
||||
e2e/playwright-report/
|
||||
e2e/test-results/
|
||||
e2e/.cache/
|
||||
e2e/.env.test
|
||||
# Playwright artifacts when run from the repo root instead of e2e/
|
||||
/test-results/
|
||||
/playwright-report/
|
||||
|
||||
# OS
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# Claude Code personal (per-user) settings — shared settings.json IS committed
|
||||
.claude/settings.local.json
|
||||
|
||||
81
Caddyfile
81
Caddyfile
@@ -1,26 +1,71 @@
|
||||
{$DOMAIN} {
|
||||
encode zstd gzip
|
||||
# Compress everything EXCEPT the SSE stream — gzip buffering delays
|
||||
# "real-time" likes/comments until the ~30s keep-alive tick.
|
||||
@compressible not path /api/v1/stream
|
||||
encode @compressible zstd gzip
|
||||
|
||||
# SvelteKit frontend — static assets with long-lived cache (content-hashed filenames)
|
||||
@hashed_assets path_regexp hashed /_app/immutable/.*\.[a-f0-9]{8,}\.(js|css|woff2)$
|
||||
header @hashed_assets Cache-Control "public, max-age=31536000, immutable"
|
||||
# Site-wide security headers (defense-in-depth). HSTS is free since Caddy
|
||||
# already terminates TLS. nosniff also covers all of /media/*.
|
||||
header {
|
||||
Strict-Transport-Security "max-age=31536000; includeSubDomains"
|
||||
X-Content-Type-Options "nosniff"
|
||||
Referrer-Policy "strict-origin-when-cross-origin"
|
||||
}
|
||||
|
||||
# Media previews and thumbnails
|
||||
@previews path /media/previews/* /media/thumbnails/*
|
||||
header @previews Cache-Control "public, max-age=3600"
|
||||
# X-Frame-Options: DENY everywhere EXCEPT the keepsake download endpoints, which
|
||||
# are navigated in a HIDDEN, SAME-ORIGIN iframe so a 404/429 can't unload the PWA
|
||||
# (see frontend/src/routes/export/+page.svelte). WebKit enforces XFO *before*
|
||||
# honouring Content-Disposition, so a blanket DENY makes the download silently do
|
||||
# nothing on iOS Safari — the app's primary platform. SAMEORIGIN still blocks
|
||||
# cross-origin framing.
|
||||
#
|
||||
# Split into two disjoint matchers rather than an override: Caddy applies the
|
||||
# FIRST header directive outermost, so it wins on write — a later, more specific
|
||||
# `header` would be silently ignored.
|
||||
@framable path /api/v1/export/zip /api/v1/export/html
|
||||
@not_framable not path /api/v1/export/zip /api/v1/export/html
|
||||
header @framable X-Frame-Options "SAMEORIGIN"
|
||||
header @not_framable X-Frame-Options "DENY"
|
||||
|
||||
# Original media files (private — only host can download)
|
||||
@originals path /media/originals/*
|
||||
header @originals Cache-Control "private, max-age=86400"
|
||||
# SvelteKit frontend — static assets with long-lived cache (content-hashed filenames)
|
||||
@hashed_assets path_regexp hashed /_app/immutable/.*\.[a-f0-9]{8,}\.(js|css|woff2)$
|
||||
header @hashed_assets Cache-Control "public, max-age=31536000, immutable"
|
||||
|
||||
# API — never cache
|
||||
@api path /api/*
|
||||
header @api Cache-Control "no-store"
|
||||
# Preview/thumbnail images. These are served by the app through a visibility-checked
|
||||
# alias (/api/v1/upload/{id}/{preview,thumbnail}) so moderation can revoke access;
|
||||
# the app serves no /media route at all, so there is no direct path to the bytes.
|
||||
# Privately cacheable for a short window (the app sets the same header; this is the
|
||||
# edge carve-out from the blanket no-store below). Kept short so a moderated image
|
||||
# stops being served to a direct-URL holder promptly.
|
||||
@media_api path /api/v1/upload/*/preview /api/v1/upload/*/thumbnail
|
||||
header @media_api Cache-Control "private, max-age=300"
|
||||
|
||||
# Route API and media requests to the Rust backend
|
||||
reverse_proxy /api/* app:3000
|
||||
reverse_proxy /media/* app:3000
|
||||
# API and health — never cache, EXCEPT the gated image routes above. A cached health
|
||||
# response would report the last known state rather than the current one.
|
||||
@api {
|
||||
path /api/* /health
|
||||
not path /api/v1/upload/*/preview /api/v1/upload/*/thumbnail
|
||||
}
|
||||
header @api Cache-Control "no-store"
|
||||
|
||||
# Everything else goes to SvelteKit frontend
|
||||
reverse_proxy frontend:3001
|
||||
# Route API and media requests to the Rust backend.
|
||||
#
|
||||
# The app serves no /media route at all (see the note in backend/src/main.rs) — media
|
||||
# bytes are reachable only through the visibility-checked /api/v1/upload aliases, so
|
||||
# /media/* forwards to a plain 404. The proxy line is kept deliberately: it means the
|
||||
# edge faithfully hands /media to the app, so if a future change ever re-introduces a
|
||||
# static media route the e2e gating specs see it here exactly as production would,
|
||||
# instead of being masked by the SvelteKit 404 page.
|
||||
reverse_proxy /api/* app:3000
|
||||
reverse_proxy /media/* app:3000
|
||||
|
||||
# The backend registers /health on its ROOT router, not under /api/v1, so it needs its
|
||||
# own line — without it the catch-all below hands /health to SvelteKit, which has no
|
||||
# such route and returns its 404 page. That made the documented post-deploy check
|
||||
# (`curl -fsS https://DOMAIN/health`) fail 100% of the time on a perfectly healthy
|
||||
# stack. e2e/Caddyfile.test has always carried this line; production never did.
|
||||
reverse_proxy /health app:3000
|
||||
|
||||
# Everything else goes to SvelteKit frontend
|
||||
reverse_proxy frontend:3001
|
||||
}
|
||||
|
||||
137
FOLLOWUPS.md
Normal file
137
FOLLOWUPS.md
Normal file
@@ -0,0 +1,137 @@
|
||||
# Follow-ups
|
||||
|
||||
Tracked work that was deferred during the multi-round UI/UX review pass.
|
||||
Each item has a clear acceptance criterion so a future pass can land it
|
||||
without re-deriving the context.
|
||||
|
||||
## A11y — assistive-tech containment inside modals
|
||||
|
||||
**Problem.** Open modals (LightboxModal, ConfirmSheet, Modal, OnboardingGuide,
|
||||
join PIN modal, account data-mode sheet, export HTML guide) trap keyboard Tab
|
||||
via `focusTrap`, but VoiceOver rotor / TalkBack arrow-key navigation can still
|
||||
escape into the page content behind the dialog. Screen-reader users hear the
|
||||
wrong context.
|
||||
|
||||
**Why deferred.** A naive sibling-walk that sets `inert` on direct children of
|
||||
the modal's parent silences the global `<Toaster>` (`aria-live="polite"` region
|
||||
mounted in `+layout.svelte`) and the `<BottomNav>` while a modal is open —
|
||||
breaking toast announcements and the visible nav state. SvelteKit has no
|
||||
built-in portal mechanism, so dialogs render inside the route tree alongside
|
||||
the Toaster.
|
||||
|
||||
**Acceptance criterion.** With any modal open:
|
||||
- VoiceOver rotor (iOS Safari) and TalkBack swipe navigation (Android Chrome)
|
||||
cannot leave the dialog subtree.
|
||||
- Toasts that fire while a modal is open are still announced.
|
||||
- Nested modals (e.g. ConfirmSheet opened from inside ContextSheet) maintain
|
||||
correct containment when the inner closes.
|
||||
|
||||
**Sketch of an approach.** One of:
|
||||
1. **Portal pattern.** Render dialogs into a dedicated `<div id="modal-root">`
|
||||
that's a sibling of the main app root in `app.html`. `focusTrap` then sets
|
||||
`inert` on the main root, leaving the modal root and the toast region (also
|
||||
moved to its own portal root) untouched.
|
||||
2. **Opt-out marker.** Walk siblings and inert them, but skip any node carrying
|
||||
a `data-modal-passthrough` attribute. Mark `<Toaster>` with it. Document
|
||||
the contract.
|
||||
3. **Stack-aware containment.** Maintain a module-level stack of open dialog
|
||||
nodes; the topmost owns the inert state, popped dialogs restore the
|
||||
previous layer. Avoids the nested-modal restoration bug.
|
||||
|
||||
Approach 1 is the cleanest long-term but the highest blast radius. Approach 2
|
||||
is the smallest patch.
|
||||
|
||||
**Files to touch.**
|
||||
- [frontend/src/lib/actions/focus-trap.ts](frontend/src/lib/actions/focus-trap.ts) — add inert logic
|
||||
- [frontend/src/lib/components/Toaster.svelte](frontend/src/lib/components/Toaster.svelte) — add passthrough marker (if approach 2) or move to a portal (if approach 1)
|
||||
- [frontend/src/app.html](frontend/src/app.html) — add `<div id="modal-root">` (if approach 1)
|
||||
|
||||
## Feed — DOM-windowing virtualization (IMPLEMENTED — residual validation owed)
|
||||
|
||||
**Status.** Implemented in [frontend/src/lib/components/VirtualFeed.svelte](frontend/src/lib/components/VirtualFeed.svelte)
|
||||
using `@tanstack/svelte-virtual`'s `createWindowVirtualizer`. Both the **list**
|
||||
view (dynamic `measureElement` heights, keyed by upload id with `anchorTo:'start'`
|
||||
so an SSE prepend doesn't yank a scrolled-down reader) and the **grid** view
|
||||
(three measured square tiles per row) now keep only the on-screen window (+overscan)
|
||||
in the DOM instead of one node per upload. The window virtualizer scrolls the
|
||||
document, so the sticky header, pull-to-refresh, the infinite-scroll sentinel and
|
||||
the bottom nav are untouched. The earlier `content-visibility` band-aid was removed
|
||||
from `FeedListCard` (it interferes with real-height measurement), and the old
|
||||
`FeedGrid.svelte` was deleted (its sole consumer migrated to `VirtualFeed`).
|
||||
|
||||
**Verified.** `svelte-check` 0 errors, production build clean. The integration
|
||||
follows the library's documented window-virtualizer contract (confirmed against
|
||||
`virtual-core` source: item `start` includes `scrollMargin`, `getTotalSize()`
|
||||
excludes it; the SSR path is guarded by `getScrollElement()` returning null).
|
||||
|
||||
**Residual validation owed (needs the running app — could not be done headless).**
|
||||
- Manual scroll testing on a ~1000-item event: confirm no jank, correct scrollbar
|
||||
size, and that an SSE `new-upload` / `feed-delta` prepend while scrolled down does
|
||||
not jump the viewport (the `anchorTo:'start'` + id-key path).
|
||||
- `scrollMargin` re-measure when grid filter chips change the header height (handled
|
||||
reactively via the `uploads`-length-driven effect, but unverified visually).
|
||||
- A new e2e spec that scrolls far down, likes an item via SSE, and asserts scroll
|
||||
position is retained — the existing suite only asserts a single card is visible,
|
||||
so it cannot catch a scroll regression.
|
||||
|
||||
**Files.**
|
||||
- [frontend/src/lib/components/VirtualFeed.svelte](frontend/src/lib/components/VirtualFeed.svelte) — new windowing component (list + grid)
|
||||
- [frontend/src/routes/feed/+page.svelte](frontend/src/routes/feed/+page.svelte) — renders `VirtualFeed` for both views
|
||||
- [frontend/src/lib/components/FeedListCard.svelte](frontend/src/lib/components/FeedListCard.svelte) — `content-visibility` removed
|
||||
|
||||
**Known limitations (surfaced in the post-commit review, left as-is — low impact).**
|
||||
- **Grid prepend reflows tiles.** Grid rows are index-keyed and pack 3 tiles each, so
|
||||
an SSE `new-upload` shifts every tile by one position; `anchorTo:'start'` can only
|
||||
anchor a scrolled grid reader when the prepend crosses a 3-item boundary (it adds a
|
||||
*row*). List view is unaffected (one id-keyed row per upload). A fix would key rows
|
||||
by the first tile's id and accept partial-row churn; not worth it for the rarer
|
||||
"new upload while browsing the grid" case.
|
||||
- **Filtered grid can auto-load the whole feed.** When a grid filter matches few items
|
||||
the `VirtualFeed` is short, so the infinite-scroll sentinel sits in-viewport and
|
||||
`loadMore()` fires until `nextCursor` is null — pulling all pages to widen the
|
||||
client-side search. This is pre-existing (the old `FeedGrid` had the same shape, and
|
||||
the empty-filter copy even says "scrolle weiter"), not a virtualization regression.
|
||||
If undesired, gate auto-load to list view or to actual user scroll.
|
||||
|
||||
## Feed — comment deletion leaves a stale live count
|
||||
|
||||
**Problem.** `add_comment` now broadcasts a fresh `comment_count` so feed clients patch
|
||||
the card in place, but `delete_comment` and `host_delete_comment`
|
||||
([backend/src/handlers/social.rs](backend/src/handlers/social.rs),
|
||||
[host.rs](backend/src/handlers/host.rs)) soft-delete without broadcasting any
|
||||
count/event. So a deletion leaves the count too high on every client until a full
|
||||
refetch (pull-to-refresh or an unrelated `upload-processed` merge). `toggle_like`
|
||||
already broadcasts on both add and remove, so likes are fine — the gap is
|
||||
comments-on-delete. Pre-existing, but the in-place-patch scheme makes it observable.
|
||||
|
||||
**Fix.** Emit a `new-comment` (or a `comment-deleted`) event carrying the refreshed
|
||||
`comment_count` from both delete paths, the same best-effort way `add_comment` does;
|
||||
the frontend `patchCount(..., 'comment_count')` handler already consumes it.
|
||||
|
||||
**Note — count ordering.** The broadcast count is read just after the (auto-committed)
|
||||
mutation, not inside it, so under concurrent likes/comments on one upload the SSE
|
||||
messages are last-write-wins. The frontend *replaces* (not increments) the count, so
|
||||
steady state is correct and self-healing; only document this if strict per-event
|
||||
ordering is ever required (then compute the count in-tx with a monotonic sequence).
|
||||
|
||||
## Feed — per-image exact CLS reservation
|
||||
|
||||
**Problem.** `FeedListCard` now reserves a default `aspect-[4/5]` box for photos so
|
||||
the card doesn't collapse to height 0 and reflow as images stream in (matching
|
||||
`Skeleton`). But no image dimensions are stored anywhere (not on `FeedUpload`, the
|
||||
`upload` table, or any migration), so the box is a uniform guess that crops to fit —
|
||||
the original is one tap away in the lightbox.
|
||||
|
||||
**Acceptance criterion.** Extract image width/height during the compression worker,
|
||||
store them on `upload`, expose them on `FeedUpload`, and have the card reserve the
|
||||
*true* aspect ratio (no crop, zero shift).
|
||||
|
||||
**Files to touch.**
|
||||
- backend: `services/compression.rs`, `models/upload.rs`, `handlers/feed.rs`, a migration
|
||||
- [frontend/src/lib/types.ts](frontend/src/lib/types.ts), [FeedListCard.svelte](frontend/src/lib/components/FeedListCard.svelte)
|
||||
|
||||
## Smaller nits, optional
|
||||
|
||||
- **Auto-submit on retried 4th digit.** [recover/+page.svelte](frontend/src/routes/recover/+page.svelte), [join/+page.svelte](frontend/src/routes/join/+page.svelte) — after a wrong PIN, deleting one digit and retyping triggers an immediate submit. Backend's 3-attempts/15-min lockout makes this safe; could feel hair-trigger after a typo. Consider gating the second auto-submit per input session behind an explicit button press.
|
||||
- **Onboarding pip tap target on the vertical axis.** [OnboardingGuide.svelte](frontend/src/lib/components/OnboardingGuide.svelte) — current `p-2.5` yields ~26 px height, meets WCAG 2.2 AA (≥24 px) but below iOS HIG / Material's 44 / 48 dp recommendation. Bumping to `p-3` is the easy improvement; further increases start crowding the row.
|
||||
- **Migrate bespoke focus-trapped dialogs to `<Modal>`.** Join PIN modal, OnboardingGuide, LightboxModal, HTML guide, account data-mode sheet — all currently roll their own shell with `focusTrap`. They're correct, just not using the canonical primitive. Migrate when `<Modal>` gains features (e.g. the inert work above) you'd want everywhere.
|
||||
143
PROJECT.md
143
PROJECT.md
@@ -42,7 +42,7 @@ A guest scans the QR code on their way in, types their name, and is immediately
|
||||
Mobile-first Progressive Web App (PWA) — accessible via browser, no app store required.
|
||||
|
||||
### Status
|
||||
Idea / Planning phase. Greenfield personal project.
|
||||
Implementation in progress (~v0.16). Core flows + new features all wired: auth, feed, upload, host/admin dashboards, ZIP + HTML-viewer export, SSE with delta-fetch on reconnect, toggleable rate limits + quotas with live per-user estimate, host PIN reset with one-time modal, data-mode (Saver/Original), Datenschutzhinweis, mobile gestures (long-press context sheet, double-tap to like), and the live Diashow with pluggable transitions. Open items: low-disk alert, event banner UI, chunked resumable upload for very large videos. See [FEATURES.md](docs/FEATURES.md) for the capability matrix.
|
||||
|
||||
---
|
||||
|
||||
@@ -208,9 +208,9 @@ Personal / private use. One event at a time. Up to ~100 users uploading ~1,000 f
|
||||
│ Axum HTTP Server (Rust — Single Binary) │
|
||||
│ │
|
||||
│ ┌─────────────┐ ┌──────────────┐ ┌────────────────────────┐ │
|
||||
│ │ REST API │ │ SSE Engine │ │ Static File Server │ │
|
||||
│ │ /api/v1/* │ │ /api/v1/ │ │ (SvelteKit build │ │
|
||||
│ │ │ │ stream │ │ output, embedded) │ │
|
||||
│ │ REST API │ │ SSE Engine │ │ Media Static Server │ │
|
||||
│ │ /api/v1/* │ │ /api/v1/ │ │ /media/* (originals, │ │
|
||||
│ │ │ │ stream │ │ previews, thumbnails) │ │
|
||||
│ └──────┬──────┘ └──────┬───────┘ └────────────────────────┘ │
|
||||
│ │ │ │
|
||||
│ ┌──────▼──────────────────────┐ ┌──────────────────────────┐ │
|
||||
@@ -245,36 +245,14 @@ Personal / private use. One event at a time. Up to ~100 users uploading ~1,000 f
|
||||
|
||||
### Docker Compose Stack
|
||||
|
||||
Four services: Postgres, the Rust API (`app`), the SvelteKit Node server (`frontend`), and Caddy. Caddy routes `/api/*` and `/media/*` to the Rust binary and everything else to the SvelteKit server. See [docker-compose.yml](docker-compose.yml) for the authoritative definition.
|
||||
|
||||
```yaml
|
||||
services:
|
||||
app:
|
||||
build: ./backend # Multi-stage Rust Dockerfile
|
||||
env_file: .env
|
||||
depends_on: [db]
|
||||
volumes:
|
||||
- media_data:/media
|
||||
restart: unless-stopped
|
||||
|
||||
db:
|
||||
image: postgres:16-alpine
|
||||
env_file: .env
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
restart: unless-stopped
|
||||
|
||||
caddy:
|
||||
image: caddy:2-alpine
|
||||
ports: ["80:80", "443:443"]
|
||||
volumes:
|
||||
- ./Caddyfile:/etc/caddy/Caddyfile:ro
|
||||
- caddy_data:/data
|
||||
depends_on: [app]
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
media_data:
|
||||
caddy_data:
|
||||
db: # postgres:16-alpine, persisted in postgres_data volume
|
||||
app: # ./backend — Rust API on :3000, mounts media_data:/media
|
||||
frontend: # ./frontend — SvelteKit (adapter-node) on :3001
|
||||
caddy: # caddy:2-alpine — terminates TLS on :80/:443, proxies app + frontend
|
||||
```
|
||||
|
||||
### Caddyfile
|
||||
@@ -345,11 +323,14 @@ COMPRESSION_WORKER_CONCURRENCY=2
|
||||
|-------|---------|---------|
|
||||
| `new-upload` | `{ id, preview_url, uploader, caption, created_at }` | Upload processing complete |
|
||||
| `new-comment` | `{ id, upload_id, body, uploader, created_at }` | Comment posted |
|
||||
| `new-like` | `{ upload_id, like_count }` | Like toggled |
|
||||
| `like-update` | `{ upload_id, like_count }` | Like toggled |
|
||||
| `upload-deleted` | `{ upload_id }` | Upload deleted |
|
||||
| `event-closed` | `{}` | Host locks uploads |
|
||||
| `event-opened` | `{}` | Host unlocks uploads |
|
||||
| `export-available` | `{ types: ["zip","html"] }` | Export generation complete |
|
||||
| `upload-processed` | `{ upload_id, preview_url, thumbnail_url }` | Server-side compression / preview generation finished |
|
||||
| `upload-error` | `{ upload_id, message }` | Compression / preview generation failed |
|
||||
| `export-progress` | `{ type, progress_pct }` | Periodic progress update from an export job |
|
||||
|
||||
**Client SSE lifecycle:** `visibilitychange: hidden` → close connection · `visible` → reconnect + delta-fetch via `GET /api/v1/feed/delta?since=`
|
||||
|
||||
@@ -372,7 +353,7 @@ COMPRESSION_WORKER_CONCURRENCY=2
|
||||
| Real-Time | Axum SSE + `tokio::sync::broadcast` | Native, lightweight, perfect for fan-out at this scale |
|
||||
| ZIP Export | `async-zip` crate | Streaming ZIP generation without buffering the full archive in RAM |
|
||||
| HTML Export | `minijinja` (Rust templating) | Generates `Memories.html` as a single self-contained file |
|
||||
| Rate Limiting | `tower-governor` | Token-bucket per IP / per user; config from DB; hot-reloadable |
|
||||
| Rate Limiting | Custom in-memory sliding-window limiter ([services/rate_limiter.rs](backend/src/services/rate_limiter.rs)) | Per IP / per user; limits read from `config` DB table on each request; hot-reloadable without restart |
|
||||
| Reverse Proxy | Caddy 2 | Automatic HTTPS via Let's Encrypt; zero certificate management |
|
||||
| Containerisation | Docker + Docker Compose | Full stack in one file; `.env` for all config; single-command deploy |
|
||||
| Infrastructure | Hetzner CX33 (4 vCPU, 8 GB RAM, 80 GB SSD, 20 TB traffic) | Well-sized; 20 TB/month means post-event bulk downloads are no issue |
|
||||
@@ -417,9 +398,9 @@ No paid third-party services required.
|
||||
|
||||
| Role | Permissions |
|
||||
|------|------------|
|
||||
| Guest | Upload (within quota), caption/hashtag, like, comment, delete own content, view feed, download export (after release) |
|
||||
| Host | All guest permissions + ban/unban users (with upload visibility prompt), delete any content, promote guests to Host, lock/unlock uploads, release gallery export |
|
||||
| Admin | All Host permissions + configure storage/file/rate limits, quota tolerance, view disk usage, manage app config, trigger export generation |
|
||||
| Guest | Upload (within quota), caption/hashtag, like, comment, delete own content, view feed, download export (after release), pick data mode, read privacy note |
|
||||
| Host | All guest permissions + ban/unban users (with upload visibility prompt), delete any content, promote guests to Host, demote *other* Hosts to guest (never self), reset guest PINs (planned), lock/unlock uploads, release gallery export |
|
||||
| Admin | All Host permissions + reset any non-admin PIN, configure storage/file/rate limits with on/off toggles, edit quota tolerance and per-area quota toggles, edit the Datenschutzhinweis, view disk usage, manage app config, trigger export generation |
|
||||
| Banned Guest | View feed only — cannot upload, like, comment, or export |
|
||||
|
||||
### Compliance
|
||||
@@ -473,37 +454,33 @@ Full-quality originals only. File naming: `{date}_{time}_{username}_{original_fi
|
||||
|
||||
### Export Type 2: HTML Offline Viewer (`Memories.zip`)
|
||||
|
||||
The HTML export is a **pre-built SvelteKit static app** (`adapter-static`, `ssr=false`) shipped together with the event data. It is a non-interactive, read-only clone of the live feed — same components, same Tailwind tokens, same look — minus auth, upload, comment, and any dashboards. Full design rationale in [docs/CONCEPT_HTML_VIEWER.md](docs/CONCEPT_HTML_VIEWER.md).
|
||||
|
||||
```
|
||||
Memories/
|
||||
Memories.html ← single entry point (all CSS + JS inlined; no external deps)
|
||||
README.txt ← plain-text setup guide (in German, as the UI language)
|
||||
Photos/ ...
|
||||
Videos/ ...
|
||||
index.html ← entry point; open this in any browser
|
||||
_app/
|
||||
immutable/... ← hashed JS/CSS bundles (viewer SPA)
|
||||
data.json ← event metadata, posts, comments, likes, hashtags
|
||||
media/
|
||||
{id}_thumb.jpg ← grid thumbnails (≈400 px wide)
|
||||
{id}_full.jpg/.mp4 ← full-size media for the lightbox
|
||||
```
|
||||
|
||||
**Fully self-contained / true offline:** `Memories.html` is a single file with all CSS and JS inlined as `<style>` and `<script>` tags — no external stylesheets, no CDN scripts, no network requests. All images and videos are referenced via **relative paths** to the sibling `Photos/` and `Videos/` folders — not base64-embedded (that would make the HTML file unworkably large). The ZIP must be unzipped first; relative paths resolve correctly from any location on disk.
|
||||
**How it works:** open `index.html` in any modern browser. The viewer hydrates client-side, `fetch('./data.json')` loads the event snapshot, all media references are relative paths into `media/`. No network calls, no service required. The ZIP must be unzipped first; the viewer does not run from inside an archive.
|
||||
|
||||
**`Memories.html` features:** responsive photo/video grid, fullscreen lightbox, client-side hashtag filter chips, comments + like counts per upload, uploader name + timestamp, warm keepsake album aesthetic — all in self-contained vanilla JS + CSS.
|
||||
**Viewer feature parity with the live app:**
|
||||
- List view (chronological) and 3-column grid view with the same toggle as the live app
|
||||
- Lightbox with swipe navigation
|
||||
- Hashtag filter chips and grid-view search/autocomplete
|
||||
- Like counts and comment lists shown as a static snapshot from export time
|
||||
- All UI strings in German
|
||||
|
||||
**`README.txt`** (in German, as the app's UI language):
|
||||
```
|
||||
Willkommen in der Event-Galerie!
|
||||
**Build flow:** The viewer lives at [frontend/export-viewer/](frontend/export-viewer/) and is built ahead of time into [backend/static/export-viewer/](backend/static/export-viewer/) (committed to the repo). The export job embeds those assets via `include_dir!`, generates `data.json` from the database, processes thumbnails/full-sized variants, and streams the ZIP.
|
||||
|
||||
So geht's:
|
||||
1. Entpacke diese ZIP-Datei
|
||||
(Windows: Rechtsklick > "Alle extrahieren"; Mac: Doppelklick;
|
||||
Handy: Dateimanager-App verwenden).
|
||||
2. Öffne die Datei "Memories.html" in deinem Browser
|
||||
(z. B. Chrome, Safari oder Firefox).
|
||||
3. Stöbere durch alle Fotos und Videos.
|
||||
Du kannst nach Hashtags filtern — klicke einfach auf einen Hashtag.
|
||||
4. Eine Internetverbindung ist nicht nötig.
|
||||
Alles ist lokal auf deinem Gerät gespeichert.
|
||||
**Source files (ZIP archive export, see below)** still contain the unmodified originals — the viewer is the polished read-only experience, the ZIP is the raw archive.
|
||||
|
||||
Viel Freude mit den Erinnerungen!
|
||||
```
|
||||
|
||||
For video-heavy events the ZIP can be several GB. The in-app download guide warns guests: *"Am besten im WLAN herunterladen."* ("Best downloaded on Wi-Fi.")
|
||||
For video-heavy events the viewer ZIP can be several GB. The in-app download guide warns guests: *"Am besten im WLAN herunterladen."* ("Best downloaded on Wi-Fi.")
|
||||
|
||||
---
|
||||
|
||||
@@ -625,7 +602,9 @@ CREATE TABLE "user" (
|
||||
recovery_pin_hash TEXT NOT NULL, -- bcrypt(PIN)
|
||||
total_upload_bytes BIGINT NOT NULL DEFAULT 0, -- running sum for quota checks
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
-- No UNIQUE(event_id, display_name) — PIN disambiguates name collisions
|
||||
-- Case-insensitive UNIQUE on (event_id, LOWER(display_name)) added by migration 007.
|
||||
-- Name collisions are rejected on join; the user is prompted to recover with their PIN
|
||||
-- (or to pick a different name).
|
||||
);
|
||||
|
||||
-- ─────────────────────────────────────────
|
||||
@@ -730,12 +709,23 @@ CREATE TABLE config (
|
||||
INSERT INTO config (key, value) VALUES
|
||||
('max_image_size_mb', '20'),
|
||||
('max_video_size_mb', '500'),
|
||||
('upload_rate_per_hour', '10'),
|
||||
('upload_rate_per_hour', '100'), -- raised from 10 in migration 015 (guests upload bursts of 10-20)
|
||||
('feed_rate_per_min', '60'),
|
||||
('export_rate_per_day', '3'),
|
||||
('quota_tolerance', '0.75'),
|
||||
('estimated_guest_count', '100'),
|
||||
('compression_concurrency', '2')
|
||||
('compression_concurrency', '2'),
|
||||
-- Planned (see docs/FEATURES.md §2.6 and §2.7):
|
||||
-- on/off switches for rate limits and quotas, and the privacy note text
|
||||
('rate_limits_enabled', 'true'),
|
||||
('upload_rate_enabled', 'true'),
|
||||
('feed_rate_enabled', 'true'),
|
||||
('export_rate_enabled', 'true'),
|
||||
('join_rate_enabled', 'true'),
|
||||
('quota_enabled', 'true'),
|
||||
('storage_quota_enabled', 'true'),
|
||||
('upload_count_quota_enabled', 'true'),
|
||||
('privacy_note', '') -- free text, whitespace + newlines preserved, no HTML
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
```
|
||||
|
||||
@@ -1143,16 +1133,19 @@ eventsnap/
|
||||
|
||||
### Backup Strategy
|
||||
|
||||
```bash
|
||||
# Daily (e.g. as a separate Compose service or cron on the VPS)
|
||||
pg_dump $DATABASE_URL | gzip > /media/backups/db_$(date +%Y-%m-%d).sql.gz
|
||||
Three artefacts in three places: the database, the `media_data` volume
|
||||
(originals + derivatives), and the **separate** `exports_data` volume. See
|
||||
[README.md](README.md#backup) for the exact commands.
|
||||
|
||||
# Weekly: rsync /media volume to Hetzner Storage Box
|
||||
rsync -az /opt/eventsnap/media/ \
|
||||
user@u123456.your-storagebox.de:backup/eventsnap/
|
||||
```
|
||||
Everything runs through `docker compose` / `docker run`, because `DATABASE_URL`
|
||||
and the `/media` and `/exports` paths only exist inside the compose network —
|
||||
they are not host paths, and `DATABASE_URL` is never exported into an operator's
|
||||
shell.
|
||||
|
||||
The `/media` volume contains originals, previews, thumbnails, generated exports, and DB backups — a single volume to back up.
|
||||
Export archives are deliberately outside `MEDIA_PATH` (`EXPORT_PATH=/exports`): a
|
||||
keepsake contains every photo in the event, and keeping it off the media tree is
|
||||
what stops it being reachable except through the ticket-gated handler. A backup
|
||||
of the media volume alone silently loses every generated keepsake.
|
||||
|
||||
---
|
||||
|
||||
@@ -1173,7 +1166,11 @@ The `/media` volume contains originals, previews, thumbnails, generated exports,
|
||||
|
||||
| Decision | Chosen | Rationale |
|
||||
|----------|--------|-----------|
|
||||
| Recovery mechanism | 4-digit PIN, stored in `localStorage` + "My Account" page | Simple for non-technical guests; no email required |
|
||||
| Recovery mechanism | 4-digit PIN, stored in `localStorage` + "My Account" page; Host/Admin can issue a fresh PIN via the user list when a guest loses it entirely | Simple for non-technical guests; no email required; Host-mediated reset preserves the no-email identity model |
|
||||
| Host demotion authority | Hosts can demote other Hosts (never themselves); Admin can demote anyone non-admin | Avoids requiring an Admin for every staffing change at the event |
|
||||
| Privacy note | Free-text, plain (no HTML), admin-edited, rendered preformatted in My Account | Many events need a per-event privacy statement; preformatted text avoids any markup-injection risk |
|
||||
| Data mode | Per-device `localStorage` setting (Saver / Original), default Saver | A guest can be on cellular on one device and Wi-Fi on another; per-device is the right scope |
|
||||
| Rate-limit & quota toggles | On/off switches plus numeric values in the `config` table | Lets the Admin disable enforcement for testing or trusted events without redeploying |
|
||||
| Admin dashboard path | `/admin` (standard route) | Correct auth checks are the security; obscure paths add no meaningful protection |
|
||||
| ZIP contents | Full-quality originals only (Photos + Videos folders) | Clean and simple; no metadata JSON |
|
||||
| HTML export assets | Fully offline (relative paths, CSS/JS inlined) | True offline experience; no external dependencies |
|
||||
@@ -1215,7 +1212,7 @@ The `/media` volume contains originals, previews, thumbnails, generated exports,
|
||||
| `uuid` | UUID v7 (time-sortable) |
|
||||
| `serde` / `serde_json` | Serialisation |
|
||||
| `tower` / `tower-http` | Middleware stack (CORS, compression, static files, request tracing) |
|
||||
| `tower-governor` | Token-bucket rate limiting (per IP and per user) |
|
||||
| (custom limiter, no crate) | Token-bucket / sliding window built in-tree at [services/rate_limiter.rs](backend/src/services/rate_limiter.rs) |
|
||||
| `tokio::sync::Semaphore` | Bounded worker pool for compression tasks |
|
||||
| `async-zip` | Streaming ZIP export (no in-memory buffer) |
|
||||
| `minijinja` | HTML export template rendering (`Memories.html`) |
|
||||
|
||||
387
README.md
387
README.md
@@ -34,7 +34,6 @@ A guest scans the QR code on their way in, types their name, and is immediately
|
||||
### Planned (v1.x)
|
||||
|
||||
- Individual file download button
|
||||
- Low-disk alert (< 10 GB free)
|
||||
- Event banner / cover image
|
||||
- Chunked resumable upload for large videos
|
||||
- Host-curated story highlights
|
||||
@@ -77,6 +76,7 @@ eventsnap/
|
||||
│ ├── svelte.config.js
|
||||
│ └── Dockerfile
|
||||
├── docker-compose.yml
|
||||
├── docker-compose.dev.yml # opt-in dev overlay (publishes Postgres on the host)
|
||||
├── Caddyfile
|
||||
└── .env.example
|
||||
```
|
||||
@@ -93,30 +93,139 @@ eventsnap/
|
||||
### Deploy on a fresh VPS
|
||||
|
||||
```bash
|
||||
# 1. Clone the repository
|
||||
git clone https://git.mc02.dev/fabi/EventSnap.git
|
||||
cd EventSnap
|
||||
# 1. Clone the repository (into a lowercase dir, matching the paths used below)
|
||||
git clone https://git.mc02.dev/fabi/EventSnap.git eventsnap
|
||||
cd eventsnap
|
||||
|
||||
# 2. Configure environment
|
||||
# 2. Configure environment — set EVERY secret NOW, before step 3.
|
||||
cp .env.example .env
|
||||
nano .env # set DOMAIN, JWT_SECRET, ADMIN_PASSWORD_HASH, EVENT_NAME, etc.
|
||||
nano .env # DOMAIN, EVENT_NAME, EVENT_SLUG,
|
||||
# JWT_SECRET, ADMIN_PASSWORD_HASH,
|
||||
# POSTGRES_PASSWORD *and* the same password inside DATABASE_URL
|
||||
# (see "Generate required secrets" below)
|
||||
|
||||
# 3. Start the stack
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
> **Set every secret before step 3 — `POSTGRES_PASSWORD` especially.** Postgres reads it **only
|
||||
> when it initialises its data directory**, which happens on the very first `docker compose up -d`.
|
||||
> Changing it in `.env` afterwards does not change the stored password: the app then authenticates
|
||||
> with the new one against a volume holding the old one, and you get a permanent restart loop with
|
||||
> `password authentication failed for user "eventsnap"`. The only fixes are restoring the old
|
||||
> password or `docker compose down -v`, which **deletes the database, the media and the exports**.
|
||||
> Getting it right once, up front, costs nothing; getting it wrong costs the volume.
|
||||
|
||||
Caddy automatically obtains a Let's Encrypt certificate on first start. The app is live at `https://DOMAIN` within ~30 seconds.
|
||||
|
||||
> **If the site never comes up:** with `APP_ENV=production` the backend **refuses to boot** while
|
||||
> `JWT_SECRET`, `ADMIN_PASSWORD_HASH` or the password inside `DATABASE_URL` still hold the
|
||||
> `.env.example` placeholders (this is deliberate — a publicly-known signing key or database
|
||||
> password is worse than downtime). Caddy then waits on the unhealthy `app` container and never
|
||||
> serves. Check `docker compose logs app` — a "Refusing to start … placeholder …" line lists
|
||||
> **every** unset secret at once, so one edit fixes them all.
|
||||
>
|
||||
> **If it comes up but keeps restarting with `password authentication failed for user
|
||||
> "eventsnap"`:** `POSTGRES_PASSWORD` was changed after the database volume was created. Postgres
|
||||
> applies that variable only at initialisation, so `.env` and the stored password have drifted
|
||||
> apart permanently. `docker compose logs app` spells this out. Before the event, with nothing
|
||||
> worth keeping:
|
||||
>
|
||||
> ```bash
|
||||
> docker compose down -v && docker compose up -d # -v DELETES db + media + exports. No undo.
|
||||
> ```
|
||||
>
|
||||
> **Once the event has real data, never do that.** Put the original password back into
|
||||
> `DATABASE_URL`, or change the stored one instead:
|
||||
>
|
||||
> ```bash
|
||||
> docker compose exec db psql -U "$POSTGRES_USER" -c \
|
||||
> "ALTER ROLE eventsnap WITH PASSWORD 'the-password-now-in-your-.env';"
|
||||
> ```
|
||||
|
||||
> **Production note:** `docker compose up -d` does **not** expose the database — Postgres is reachable only on the internal Docker network. For local development where you need host access to Postgres, opt into the dev overlay explicitly:
|
||||
> ```bash
|
||||
> docker compose -f docker-compose.yml -f docker-compose.dev.yml up
|
||||
> ```
|
||||
|
||||
### Updating an existing deployment
|
||||
|
||||
> **`docker compose up -d` alone will NOT deploy your changes.** `app` and `frontend` are
|
||||
> `build:` services with no published image tag, and Compose has no source-change detection:
|
||||
> if an image with that name already exists it is reused. After a `git pull` the command
|
||||
> reports `Container … Running`, changes nothing, and **exits 0** — so a deploy that shipped
|
||||
> nothing looks exactly like a successful one. `--build` is what makes it real.
|
||||
|
||||
```bash
|
||||
cd /path/to/eventsnap
|
||||
|
||||
# 1. Back up first — migrations run automatically on boot and are not reversible in place.
|
||||
# (See "Backup" below; the database dump is the one that matters here.)
|
||||
|
||||
# 2. Fetch the new code.
|
||||
git pull
|
||||
|
||||
# 3. Rebuild and restart the application services. --build is NOT optional.
|
||||
docker compose up -d --build
|
||||
|
||||
# 4. Apply any Caddyfile change. Step 3 does NOT do this — see the warning below.
|
||||
docker compose up -d --force-recreate caddy
|
||||
|
||||
# 5. Confirm the app came back up. Anything other than "ok" means check the logs.
|
||||
curl -fsS https://DOMAIN/health && echo
|
||||
|
||||
# 6. Confirm a NEW image was actually built. Note the IMAGE ID before you start and
|
||||
# compare — it must have changed. (Ignore the CREATED column; it reports the base
|
||||
# layer's age, not this build's.) An unchanged ID means step 3 ran without --build
|
||||
# and you are still serving the old code.
|
||||
docker compose images app frontend
|
||||
```
|
||||
|
||||
Migrations are applied by the backend on startup, so step 3 covers them. If `app` stays
|
||||
unhealthy afterwards, `docker compose logs app` will name the failing migration — and note
|
||||
that a migration applied by a *newer* build is not removed by checking out an older commit,
|
||||
so rolling back code without restoring the database snapshot from step 1 leaves the schema
|
||||
ahead of the binary and the app refusing to boot.
|
||||
|
||||
> **Why step 4 exists.** `--build` only rebuilds services that have a `build:` section, and
|
||||
> `caddy` is a pinned upstream image. Compose decides whether to recreate a container from its
|
||||
> *config hash*, which covers the mount **specification** (`./Caddyfile:/etc/caddy/Caddyfile:ro`)
|
||||
> but **not the file's contents** — so a `git pull` that changes `./Caddyfile` produces no
|
||||
> delta, Compose reports `Running`, and Caddy keeps serving its old config indefinitely. Exit
|
||||
> code 0 throughout.
|
||||
>
|
||||
> That is not hypothetical: the fix that made the keepsake download work on iOS
|
||||
> (`137c4ee`) touched the Caddyfile and four e2e files and nothing else, so **all** of its
|
||||
> production effect lives in that one file. Without step 4 you deploy it, watch both image IDs
|
||||
> change, and iOS downloads stay broken.
|
||||
>
|
||||
> `--force-recreate` rather than `restart` or `caddy reload`: the bind mount is resolved to an
|
||||
> **inode** when the container is created, and `git pull` replaces the file instead of editing
|
||||
> it in place, so the container can still be bound to the old, now-unlinked inode. A restart
|
||||
> then re-reads the stale content. Recreating the container re-resolves the path.
|
||||
|
||||
`db` is never touched, and recreating `caddy` does not disturb the `caddy_data` volume, so the
|
||||
TLS certificate and all data volumes survive.
|
||||
|
||||
### Generate required secrets
|
||||
|
||||
```bash
|
||||
# JWT secret (64 random bytes)
|
||||
openssl rand -hex 64
|
||||
|
||||
# Admin password hash (bcrypt, cost 12)
|
||||
htpasswd -bnBC 12 "" yourpassword | tr -d ':\n'
|
||||
# Database password (goes in BOTH DATABASE_URL and POSTGRES_PASSWORD)
|
||||
openssl rand -hex 24
|
||||
|
||||
# Admin password hash (bcrypt). Uses an image the stack already pulls, so it needs
|
||||
# nothing installed on the host — `htpasswd` lives in apache2-utils, which a stock
|
||||
# VPS does not have. Emits cost 14 rather than 12; that is fine (admin login is
|
||||
# rate-limited and hashed off the async runtime), and any $2a/$2b/$2y hash verifies.
|
||||
docker run --rm caddy:2-alpine caddy hash-password --plaintext 'yourpassword'
|
||||
```
|
||||
|
||||
Wrap the resulting hash in **single quotes** in `.env` — see the note there; a bcrypt
|
||||
hash is full of `$`, and both Compose and dotenvy would otherwise eat those segments.
|
||||
|
||||
### Environment Variables
|
||||
|
||||
See [.env.example](.env.example) for the full list with descriptions and defaults. Key variables:
|
||||
@@ -144,51 +253,269 @@ See [.env.example](.env.example) for the full list with descriptions and default
|
||||
┌───▼────┐ ┌─────▼──────┐
|
||||
│ app │ │ frontend │
|
||||
│ :3000 │ │ :3001 │
|
||||
│ (Rust) │ │ (SvelteKit)│
|
||||
│ (Rust) │ │(SvelteKit) │
|
||||
└───┬────┘ └────────────┘
|
||||
│
|
||||
┌───▼────┐
|
||||
│ db │
|
||||
│ :5432 │
|
||||
│(Postgres│
|
||||
│(Postgres)│
|
||||
└────────┘
|
||||
```
|
||||
|
||||
- `/api/*` and `/media/*` → Rust backend
|
||||
- Everything else → SvelteKit frontend
|
||||
- Named volumes: `postgres_data`, `media_data`, `caddy_data`
|
||||
- `/api/*` → Rust backend
|
||||
- Everything else → SvelteKit frontend (`adapter-node`)
|
||||
- Named volumes: `postgres_data`, `media_data`, `exports_data`, `caddy_data`
|
||||
|
||||
Media is **not** served as static files. Every image goes through a
|
||||
visibility-checked alias (`/api/v1/upload/{id}/{preview,display,thumbnail,original}`)
|
||||
so a host takedown or a ban actually revokes access to the bytes.
|
||||
|
||||
---
|
||||
|
||||
## Sizing the disk
|
||||
|
||||
`postgres_data`, `media_data` and `exports_data` are all Docker named volumes under
|
||||
`/var/lib/docker/volumes`, so **they share one filesystem**. Filling it does not
|
||||
degrade one subsystem — Postgres stops being able to write and the whole event goes
|
||||
down.
|
||||
|
||||
Uploads are self-limiting. `per_user_limit = free_disk × quota_tolerance ÷
|
||||
active_uploaders` is recomputed against live free space on every upload, so guests
|
||||
converge on a fixed point at `tolerance / (1 + tolerance)` of the free space you
|
||||
started with — **43%** at the default 0.75. On an 80 GB box with ~70 GB free after
|
||||
the OS and images, media settles at ~30 GB and stops.
|
||||
|
||||
**The keepsake is what the 80 GB baseline does not cover.** `Gallery.zip` and
|
||||
`Memories.zip` are built concurrently and each is roughly a second copy of every
|
||||
original: both write their media `Compression::Stored`, and `Memories.zip` streams the
|
||||
untouched original for every video and for every image at or under 5 MB. So a release
|
||||
wants room for **two more copies of the gallery** on top of the gallery itself.
|
||||
|
||||
| Stage | Used | Free (80 GB box) |
|
||||
|---|---|---|
|
||||
| Fresh box (OS + images) | ~10 GB | ~70 GB |
|
||||
| Guests reach the quota fixed point | ~40 GB | ~40 GB |
|
||||
| Host releases → both archives | ~100 GB | **ENOSPC** |
|
||||
|
||||
Two ways to size for it:
|
||||
|
||||
- **Provision ~3× your expected media** on one volume (media + two archives), or
|
||||
- **give `exports_data` its own volume** so a full export cannot reach Postgres, and
|
||||
size that one at ~2× expected media.
|
||||
|
||||
This is no longer silent. The export refuses up front with the two numbers rather than
|
||||
hitting ENOSPC halfway through a multi-GB write, a rebuild reclaims the superseded
|
||||
generation before it starts (so peak is one generation, not two), and the host
|
||||
dashboard warns as soon as the keepsake would not fit — which is the only point at
|
||||
which anyone can still do something about it.
|
||||
|
||||
---
|
||||
|
||||
## Backup
|
||||
|
||||
```bash
|
||||
# Database snapshot
|
||||
pg_dump $DATABASE_URL | gzip > /media/backups/db_$(date +%Y-%m-%d).sql.gz
|
||||
There are **three** things to back up, and they live in three different places.
|
||||
`DATABASE_URL` and the container paths (`/media`, `/exports`) are meaningful only
|
||||
*inside* the compose network — they are not host paths, and `DATABASE_URL` is
|
||||
never exported into an operator's shell — so every command below runs through
|
||||
`docker compose` from the repo directory.
|
||||
|
||||
# Weekly offsite sync (Hetzner Storage Box or similar)
|
||||
rsync -az /opt/eventsnap/media/ user@storagebox.example.com:backup/eventsnap/
|
||||
```bash
|
||||
# 1. Database snapshot. Runs pg_dump inside the db container (the app image has no
|
||||
# postgres client), reading credentials from the compose environment.
|
||||
# --clean --if-exists makes the dump SELF-CLEANING: without it the restore below
|
||||
# aborts on the first "already exists" against a database that has ever booted,
|
||||
# which is every database you would actually want to restore over.
|
||||
mkdir -p ./backups
|
||||
docker compose exec -T db \
|
||||
sh -c 'pg_dump --clean --if-exists -U "$POSTGRES_USER" "$POSTGRES_DB"' \
|
||||
| gzip > ./backups/db_$(date +%Y-%m-%d).sql.gz
|
||||
|
||||
# 2. Uploaded media (originals + derivatives) out of the named volume.
|
||||
# NOTE the mountpoint is /src, not /media: if the volume is ever empty, Docker
|
||||
# pre-populates a fresh mount from the image's own directory, and alpine ships a
|
||||
# /media containing cdrom/floppy/usb. Mounting somewhere the image has nothing
|
||||
# avoids silently tarring (and polluting the volume with) those.
|
||||
docker run --rm \
|
||||
-v eventsnap_media_data:/src:ro -v "$PWD/backups":/backup \
|
||||
alpine tar czf /backup/media_$(date +%Y-%m-%d).tar.gz -C /src .
|
||||
|
||||
# 3. Export archives — a SEPARATE volume (see the security note below).
|
||||
docker run --rm \
|
||||
-v eventsnap_exports_data:/src:ro -v "$PWD/backups":/backup \
|
||||
alpine tar czf /backup/exports_$(date +%Y-%m-%d).tar.gz -C /src .
|
||||
|
||||
# Offsite sync of the three artefacts above.
|
||||
rsync -az ./backups/ user@storagebox.example.com:backup/eventsnap/
|
||||
```
|
||||
|
||||
The `/media` volume holds originals, previews, thumbnails, exports, and DB backups — a single path to back up.
|
||||
Volume names are prefixed with the compose project name — `eventsnap_` if you run
|
||||
from a directory called `eventsnap`. Confirm yours with `docker volume ls`.
|
||||
|
||||
> **Exports are deliberately NOT under `/media`.** They live on their own
|
||||
> `exports_data` volume (`EXPORT_PATH=/exports`) because a keepsake archive
|
||||
> contains every photo in the event; keeping it outside the media tree is what
|
||||
> stops it being reachable except through the ticket-gated download handler.
|
||||
> Backing up only the media volume therefore loses every generated keepsake.
|
||||
|
||||
### When to run it
|
||||
|
||||
**A nightly cron is the wrong shape for this app.** Every irreplaceable byte is
|
||||
created inside one eight-hour window, and nobody can retake a wedding. Run the three
|
||||
commands above:
|
||||
|
||||
1. **The night of the event**, once uploads have stopped. This is the backup that
|
||||
matters; everything else is a formality.
|
||||
2. **After the host releases the gallery**, so the generated keepsake is captured too.
|
||||
3. Weekly thereafter, until the event is archived and torn down.
|
||||
|
||||
Take the DB dump and the media tarball **back to back**, without uploads in flight
|
||||
between them. Upload rows reference files by path — a database from 22:00 and a media
|
||||
volume from 23:00 gives you rows pointing at files the dump doesn't know about, and
|
||||
rows whose files aren't in the tarball. Locking uploads from the host dashboard first
|
||||
(**Uploads sperren**) makes the pair genuinely consistent.
|
||||
|
||||
---
|
||||
|
||||
## Restore
|
||||
|
||||
An untested backup is not a backup. Run this once against a scratch host **before**
|
||||
the event — it is roughly ten minutes, and it is the only way to find out that your
|
||||
tarball is empty or your dump is truncated while that is still a small problem.
|
||||
|
||||
```bash
|
||||
# 0. Stop the app FIRST. Migrations run on boot and a live pool will fight the
|
||||
# restore — a booting app against a half-restored schema can leave the migration
|
||||
# table and the schema disagreeing, which is its own recovery problem.
|
||||
# Leave `db` running: the dump is restored through it.
|
||||
docker compose stop app caddy
|
||||
|
||||
# 1. Database. The dump carries its own DROPs (step 1 of Backup), so this replaces
|
||||
# rather than collides. A dump taken WITHOUT --clean --if-exists will abort here
|
||||
# on the first "already exists" — restore that one into a fresh empty database
|
||||
# instead.
|
||||
gunzip -c ./backups/db_2026-07-29.sql.gz \
|
||||
| docker compose exec -T db \
|
||||
sh -c 'psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" --set ON_ERROR_STOP=1'
|
||||
|
||||
# 2. Media. NOTE the `--numeric-owner` and the chown: the app runs as a
|
||||
# NON-ROOT user (uid 100, gid 101 — `addgroup -S app && adduser -S app`), and a
|
||||
# restore that lands root-owned files makes every upload fail with EACCES deep in
|
||||
# the write path, surfacing to the guest as a generic 500 with nothing in the UI
|
||||
# to suggest permissions. The explicit chown is what guarantees it — BusyBox tar
|
||||
# (which is what `alpine` ships) has no --same-owner, and restores ownership only
|
||||
# because it runs as root here.
|
||||
docker run --rm \
|
||||
-v eventsnap_media_data:/dst -v "$PWD/backups":/backup:ro \
|
||||
alpine sh -c 'tar xzf /backup/media_2026-07-29.tar.gz -C /dst \
|
||||
--numeric-owner && chown -R 100:101 /dst'
|
||||
|
||||
# 3. Exports. Same volume-name caveat, same ownership rules.
|
||||
docker run --rm \
|
||||
-v eventsnap_exports_data:/dst -v "$PWD/backups":/backup:ro \
|
||||
alpine sh -c 'tar xzf /backup/exports_2026-07-29.tar.gz -C /dst \
|
||||
--numeric-owner && chown -R 100:101 /dst'
|
||||
|
||||
# 4. Back up. Migrations run, then export recovery re-arms any keepsake whose file
|
||||
# didn't come back with the volume.
|
||||
docker compose up -d app caddy
|
||||
docker compose logs -f app # watch for "migrations applied"
|
||||
|
||||
# 5. Verify — all three, not just the first.
|
||||
curl -fsS https://DOMAIN/health && echo # → ok
|
||||
# … then sign in as host and confirm the feed renders images (proves the media
|
||||
# volume restored AND is readable by uid 100), and that the keepsake downloads.
|
||||
```
|
||||
|
||||
If the media volume restored but images 404 while the feed lists them, the paths are
|
||||
there and the bytes aren't — check `docker compose exec app ls -ln /media/originals`
|
||||
and confirm both the files and the `100:101` ownership.
|
||||
|
||||
The restore is deliberately **not** automated. It is rare, destructive, and the one
|
||||
operation where a script that half-works is worse than a checklist someone reads.
|
||||
|
||||
---
|
||||
|
||||
## Running the backend test suite
|
||||
|
||||
```bash
|
||||
cd backend
|
||||
|
||||
# The DB-backed integration tests (backend/tests/) need a live Postgres. `#[sqlx::test]` creates a
|
||||
# throwaway database per test and runs backend/migrations/ into it — it does NOT touch this one's data.
|
||||
docker run -d --name eventsnap-test-pg -p 55433:5432 \
|
||||
-e POSTGRES_PASSWORD=postgres -e POSTGRES_DB=eventsnap postgres:16-alpine
|
||||
|
||||
export DATABASE_URL=postgres://postgres:postgres@localhost:55433/eventsnap
|
||||
cargo test # 44 unit + 12 DB-backed
|
||||
cargo clippy --all-targets -- -D warnings
|
||||
```
|
||||
|
||||
**`cargo test` requires `DATABASE_URL`** — without it the integration tests panic rather than skip.
|
||||
That is deliberate. The riskiest code in this repo is SQL (the export epoch state machine, the
|
||||
atomic quota increment, the `FOR SHARE` upload lock), and for a long time *not one line of it* was
|
||||
executed by `cargo test` — every backend test was a pure-function test, so the tests clustered
|
||||
tightly around the code that could not break and stopped exactly where it started to. Tests that
|
||||
silently skip when the database is absent recreate that hole; they were meant to be a gate.
|
||||
|
||||
## Running the E2E test suite
|
||||
|
||||
Playwright-based end-to-end tests live in [`e2e/`](e2e/). They spin up an isolated docker-compose stack (Postgres on `:55432`, Caddy on `:3101`) and exercise the SvelteKit frontend against the real Rust backend with rate limits disabled.
|
||||
|
||||
```bash
|
||||
cd e2e
|
||||
npm install
|
||||
npm run install:browsers # one-time
|
||||
npm run stack:up # bring up the test stack
|
||||
npm run test:e2e # full Phase 1 suite on chromium-desktop
|
||||
npm run test:e2e:smoke # cross-UA matrix (chromium, samsung-internet, webkit, firefox, …)
|
||||
npm run stack:down # tear it down
|
||||
```
|
||||
|
||||
See [`e2e/README.md`](e2e/README.md) for the full UA matrix, Samsung Internet escalation tiers, and the Phase 2/3 roadmap.
|
||||
|
||||
CI runs this on every PR — see [`.github/workflows/e2e.yml`](.github/workflows/e2e.yml) (desktop **and**
|
||||
mobile projects), plus [`checks.yml`](.github/workflows/checks.yml) for `cargo test`/clippy, the frontend
|
||||
unit tests, svelte-check and the e2e typecheck, and [`audit.yml`](.github/workflows/audit.yml) for
|
||||
dependency advisories.
|
||||
|
||||
**Playwright runs with `retries: 0`, including in CI.** This repo's real bugs are races, and from the
|
||||
outside a race is indistinguishable from a flake — so a retry silently resolves that ambiguity in
|
||||
favour of "flake" every time. A flake here is a bug report; treat it as one.
|
||||
|
||||
---
|
||||
|
||||
## Development Roadmap
|
||||
|
||||
Done:
|
||||
- [x] Project blueprint & architecture
|
||||
- [x] Monorepo scaffold (`backend/`, `frontend/`, Docker Compose)
|
||||
- [ ] DB schema + SQLx migrations
|
||||
- [ ] Auth flow (join, JWT, PIN recovery)
|
||||
- [ ] Upload pipeline (multipart → compression worker → SSE broadcast)
|
||||
- [ ] Client upload queue (IndexedDB, progress, retry)
|
||||
- [ ] Gallery feed (grid, SSE, hashtag filters)
|
||||
- [ ] Camera capture (`getUserMedia`)
|
||||
- [ ] Host Dashboard
|
||||
- [ ] Admin Dashboard
|
||||
- [ ] Export engine (ZIP + offline HTML)
|
||||
- [ ] Rate limiting middleware
|
||||
- [ ] End-to-end test event (10+ real devices)
|
||||
- [x] DB schema + SQLx migrations (8 migrations through compression status + case-insensitive unique names)
|
||||
- [x] Auth flow (join, JWT, 4-digit PIN with bcrypt + 3-attempt/15-min lockout, admin login)
|
||||
- [x] Upload pipeline (multipart → compression worker via `tokio::sync::Semaphore` → SSE broadcast)
|
||||
- [x] Client upload queue (IndexedDB, progress, retry, rate-limit auto-resume)
|
||||
- [x] Gallery feed (list + grid toggle, SSE live updates, hashtag chips, in-memory search + autocomplete)
|
||||
- [x] Camera capture (`getUserMedia` with front/back toggle, photo + `MediaRecorder` video)
|
||||
- [x] Host Dashboard (event lock, gallery release, ban modal with hide-uploads choice, promote/demote, user search)
|
||||
- [x] Admin Dashboard with inner tabs (Stats, Config, Export, Nutzer)
|
||||
- [x] Export engine: streaming ZIP + SvelteKit-static HTML viewer (see [docs/CONCEPT_HTML_VIEWER.md](docs/CONCEPT_HTML_VIEWER.md))
|
||||
- [x] Custom rate limiter (per-endpoint, hot-reloadable from `config` table)
|
||||
- [x] Mobile-first redesign (bottom nav + FAB, see [docs/CONCEPT_MOBILE_UI.md](docs/CONCEPT_MOBILE_UI.md))
|
||||
|
||||
Open:
|
||||
- [ ] Dynamic per-user storage quota enforcement (formula in [PROJECT.md §12](PROJECT.md); only tracking exists today)
|
||||
- [ ] Own-upload deletion UI in the lightbox (backend route exists)
|
||||
- [ ] SSE delta-fetch on foreground reconnect (scaffolded in [sse.ts](frontend/src/lib/sse.ts), not wired)
|
||||
- [ ] Live diashow / slideshow mode — see [docs/CONCEPT_DIASHOW.md](docs/CONCEPT_DIASHOW.md)
|
||||
- [ ] Individual file download button per post
|
||||
- [x] Low-disk alert — host dashboard warns below 10 GB free, or whenever the keepsake would not fit
|
||||
- [ ] Event banner / cover image
|
||||
- [ ] Chunked resumable upload for files > 100 MB
|
||||
- [ ] Shared Tailwind config between main app and export-viewer
|
||||
- [ ] End-to-end test event (10+ real devices on cellular)
|
||||
|
||||
See [docs/FEATURES.md](docs/FEATURES.md) for the up-to-date capability matrix by role.
|
||||
Speculative / v2+ ideas live in [docs/IDEAS.md](docs/IDEAS.md).
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -16,45 +16,47 @@ Please test each step in order and report any errors (console errors, wrong text
|
||||
9. ✅ Expected: Overlay disappears
|
||||
|
||||
### Step 3 — Feed & navigation
|
||||
10. ✅ Expected: Feed shows "Noch keine Fotos." empty state with an upload button
|
||||
11. ✅ Expected: Top-right has an **upload button** (blue) and a **person icon** link
|
||||
10. ✅ Expected: Feed shows the empty state ("Noch keine Fotos." or similar) with a hint to upload
|
||||
11. ✅ Expected: A **persistent bottom nav** is visible with three slots — 🏠 **Feed** on the left, an elevated 📷+ **FAB** in the center, 👤 **Account** on the right
|
||||
|
||||
### Step 4 — My Account page
|
||||
12. Click the **person icon** in the top-right
|
||||
12. Tap the **👤 Account** tab in the bottom nav
|
||||
13. ✅ Expected: `/account` page shows your name (`Max`), a blue "Gast" badge, session expiry date, and your PIN displayed large in an amber box
|
||||
14. Click **Kopieren** — check clipboard contains your PIN
|
||||
14. Tap **Kopieren** — check the clipboard contains your PIN
|
||||
15. ✅ Expected: Button briefly shows "Kopiert!"
|
||||
16. Click **Zur Galerie** to go back to the feed
|
||||
16. Tap the 🏠 **Feed** tab to go back
|
||||
|
||||
### Step 5 — Upload
|
||||
17. Click **Hochladen** — this takes you to `/upload`
|
||||
18. Try uploading a photo from your device library
|
||||
19. ✅ Expected: Photo appears in queue with a progress bar, then completes
|
||||
20. Go back to `/feed` — ✅ Expected: your photo appears in the feed grid
|
||||
17. Tap the central **📷+ FAB** in the bottom nav
|
||||
18. ✅ Expected: A bottom sheet slides up offering **Kamera** and **Galerie** options
|
||||
19. Tap **Galerie** → pick a photo from your device library
|
||||
20. ✅ Expected: Preview screen (`/upload`) shows the staged file with an optional caption / hashtag editor
|
||||
21. Tap **Hochladen**
|
||||
22. ✅ Expected: You return to the feed immediately; the FAB shows a small badge while uploading; the photo appears in the feed once processing completes
|
||||
|
||||
### Step 6 — Onboarding guide not shown again
|
||||
21. Reload the page at `/feed`
|
||||
22. ✅ Expected: The onboarding overlay does **not** appear (already dismissed)
|
||||
23. Reload the page at `/feed`
|
||||
24. ✅ Expected: The onboarding overlay does **not** appear (already dismissed)
|
||||
|
||||
### Step 7 — Recover (open a private/incognito window)
|
||||
23. Open a new **private/incognito** window at **http://localhost:5173/recover**
|
||||
24. Enter the same name (`Max`) and the PIN you copied
|
||||
25. ✅ Expected: You're redirected to the feed with the same account
|
||||
25. Open a new **private/incognito** window at **http://localhost:5173/recover**
|
||||
26. Enter the same name (`Max`) and the PIN you copied
|
||||
27. ✅ Expected: You're redirected to the feed with the same account
|
||||
|
||||
### Step 8 — Upload rate-limit auto-retry
|
||||
26. Upload more than 20 photos in one hour to trigger the rate limit
|
||||
27. ✅ Expected: When the limit is hit, remaining items stay **Wartend** (not error)
|
||||
28. ✅ Expected: An amber banner appears in the queue: "Upload-Limit erreicht. Wird in Xs automatisch fortgesetzt."
|
||||
29. ✅ Expected: The countdown ticks down and uploads resume automatically when it reaches 0
|
||||
28. Upload more than the per-hour limit of photos in quick succession to trigger the rate limit
|
||||
29. ✅ Expected: When the limit is hit, remaining items stay **Wartend** (not error)
|
||||
30. ✅ Expected: An amber banner appears in the queue: "Upload-Limit erreicht. Wird in Xs automatisch fortgesetzt."
|
||||
31. ✅ Expected: The countdown ticks down and uploads resume automatically when it reaches 0
|
||||
|
||||
### Step 9 — Name uniqueness (case-insensitive)
|
||||
30. In a private/incognito window go to **http://localhost:5173/join**
|
||||
31. Enter `max` or `MAX` — the same name already taken in Step 1 (different case)
|
||||
32. ✅ Expected: Instead of creating a new account, an amber warning appears: „Max ist bereits vergeben." with name tips
|
||||
33. ✅ Expected: A PIN input and **Anmelden** button appear, plus an **Anderen Namen wählen** button
|
||||
34. Enter your PIN from Step 1 and click **Anmelden**
|
||||
35. ✅ Expected: You're signed in to the existing `Max` account and redirected to the feed
|
||||
36. Alternatively, click **Anderen Namen wählen** — ✅ Expected: the name input reappears with `max` pre-filled so you can edit it
|
||||
32. In a private/incognito window go to **http://localhost:5173/join**
|
||||
33. Enter `max` or `MAX` — the same name already taken in Step 1 (different case)
|
||||
34. ✅ Expected: Instead of creating a new account, an amber warning appears: „Max ist bereits vergeben." with name tips
|
||||
35. ✅ Expected: A PIN input and **Anmelden** button appear, plus an **Anderen Namen wählen** button
|
||||
36. Enter your PIN from Step 1 and click **Anmelden**
|
||||
37. ✅ Expected: You're signed in to the existing `Max` account and redirected to the feed
|
||||
38. Alternatively, click **Anderen Namen wählen** — ✅ Expected: the name input reappears with `max` pre-filled so you can edit it
|
||||
|
||||
---
|
||||
|
||||
|
||||
219
backend/Cargo.lock
generated
219
backend/Cargo.lock
generated
@@ -65,56 +65,6 @@ dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstream"
|
||||
version = "1.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d"
|
||||
dependencies = [
|
||||
"anstyle",
|
||||
"anstyle-parse",
|
||||
"anstyle-query",
|
||||
"anstyle-wincon",
|
||||
"colorchoice",
|
||||
"is_terminal_polyfill",
|
||||
"utf8parse",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstyle"
|
||||
version = "1.0.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000"
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-parse"
|
||||
version = "1.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e"
|
||||
dependencies = [
|
||||
"utf8parse",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-query"
|
||||
version = "1.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc"
|
||||
dependencies = [
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-wincon"
|
||||
version = "3.0.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d"
|
||||
dependencies = [
|
||||
"anstyle",
|
||||
"once_cell_polyfill",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.102"
|
||||
@@ -513,6 +463,17 @@ dependencies = [
|
||||
"shlex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfb"
|
||||
version = "0.7.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d38f2da7a0a2c4ccf0065be06397cc26a81f4e528be095826eee9d4adbb8c60f"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"fnv",
|
||||
"uuid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.4"
|
||||
@@ -543,46 +504,12 @@ dependencies = [
|
||||
"inout",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "clap"
|
||||
version = "4.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b193af5b67834b676abd72466a96c1024e6a6ad978a1f484bd90b85c94041351"
|
||||
dependencies = [
|
||||
"clap_builder",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "clap_builder"
|
||||
version = "4.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
"clap_lex",
|
||||
"strsim",
|
||||
"terminal_size",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "clap_lex"
|
||||
version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9"
|
||||
|
||||
[[package]]
|
||||
name = "color_quant"
|
||||
version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
|
||||
|
||||
[[package]]
|
||||
name = "colorchoice"
|
||||
version = "1.0.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570"
|
||||
|
||||
[[package]]
|
||||
name = "compression-codecs"
|
||||
version = "0.4.37"
|
||||
@@ -666,15 +593,6 @@ dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-channel"
|
||||
version = "0.5.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2"
|
||||
dependencies = [
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-deque"
|
||||
version = "0.8.6"
|
||||
@@ -805,27 +723,6 @@ dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "env_filter"
|
||||
version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32e90c2accc4b07a8456ea0debdc2e7587bdd890680d71173a15d4ae604f6eef"
|
||||
dependencies = [
|
||||
"log",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "env_logger"
|
||||
version = "0.11.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0621c04f2196ac3f488dd583365b9c09be011a4ab8b9f37248ffcc8f6198b56a"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
"env_filter",
|
||||
"log",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "equator"
|
||||
version = "0.4.2"
|
||||
@@ -897,6 +794,7 @@ dependencies = [
|
||||
"futures",
|
||||
"image",
|
||||
"include_dir",
|
||||
"infer",
|
||||
"jsonwebtoken",
|
||||
"oxipng",
|
||||
"rand 0.9.2",
|
||||
@@ -1004,6 +902,12 @@ dependencies = [
|
||||
"spin",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fnv"
|
||||
version = "1.0.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1"
|
||||
|
||||
[[package]]
|
||||
name = "foldhash"
|
||||
version = "0.1.5"
|
||||
@@ -1211,12 +1115,6 @@ dependencies = [
|
||||
"weezl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "glob"
|
||||
version = "0.3.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280"
|
||||
|
||||
[[package]]
|
||||
name = "governor"
|
||||
version = "0.6.3"
|
||||
@@ -1604,11 +1502,19 @@ checksum = "7714e70437a7dc3ac8eb7e6f8df75fd8eb422675fc7678aff7364301092b1017"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"hashbrown 0.16.1",
|
||||
"rayon",
|
||||
"serde",
|
||||
"serde_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "infer"
|
||||
version = "0.15.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cb33622da908807a06f9513c19b3c1ad50fab3e4137d82a78107d502075aa199"
|
||||
dependencies = [
|
||||
"cfb",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "inout"
|
||||
version = "0.1.4"
|
||||
@@ -1629,12 +1535,6 @@ dependencies = [
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "is_terminal_polyfill"
|
||||
version = "1.70.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695"
|
||||
|
||||
[[package]]
|
||||
name = "itertools"
|
||||
version = "0.14.0"
|
||||
@@ -1768,12 +1668,6 @@ dependencies = [
|
||||
"vcpkg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "linux-raw-sys"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
|
||||
|
||||
[[package]]
|
||||
name = "litemap"
|
||||
version = "0.8.1"
|
||||
@@ -2062,12 +1956,6 @@ version = "1.21.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "once_cell_polyfill"
|
||||
version = "1.70.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "oxipng"
|
||||
version = "9.1.5"
|
||||
@@ -2075,18 +1963,12 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "26c613f0f566526a647c7473f6a8556dbce22c91b13485ee4b4ec7ab648e4973"
|
||||
dependencies = [
|
||||
"bitvec",
|
||||
"clap",
|
||||
"crossbeam-channel",
|
||||
"env_logger",
|
||||
"filetime",
|
||||
"glob",
|
||||
"indexmap",
|
||||
"libdeflater",
|
||||
"log",
|
||||
"rayon",
|
||||
"rgb",
|
||||
"rustc-hash",
|
||||
"zopfli",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2580,19 +2462,6 @@ version = "2.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe"
|
||||
|
||||
[[package]]
|
||||
name = "rustix"
|
||||
version = "1.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustversion"
|
||||
version = "1.0.22"
|
||||
@@ -3033,12 +2902,6 @@ dependencies = [
|
||||
"unicode-properties",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "strsim"
|
||||
version = "0.11.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
|
||||
|
||||
[[package]]
|
||||
name = "subtle"
|
||||
version = "2.6.1"
|
||||
@@ -3093,16 +2956,6 @@ version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "55937e1799185b12863d447f42597ed69d9928686b8d88a1df17376a097d8369"
|
||||
|
||||
[[package]]
|
||||
name = "terminal_size"
|
||||
version = "0.4.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "230a1b821ccbd75b185820a1f1ff7b14d21da1e442e22c0863ea5f08771a8874"
|
||||
dependencies = [
|
||||
"rustix",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror"
|
||||
version = "1.0.69"
|
||||
@@ -3478,12 +3331,6 @@ version = "1.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be"
|
||||
|
||||
[[package]]
|
||||
name = "utf8parse"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
|
||||
|
||||
[[package]]
|
||||
name = "uuid"
|
||||
version = "1.23.0"
|
||||
@@ -4160,18 +4007,6 @@ version = "1.0.21"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa"
|
||||
|
||||
[[package]]
|
||||
name = "zopfli"
|
||||
version = "0.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"crc32fast",
|
||||
"log",
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zstd"
|
||||
version = "0.13.3"
|
||||
|
||||
@@ -27,9 +27,20 @@ tracing-subscriber = { version = "0.3", features = ["env-filter"] }
|
||||
dotenvy = "0.15"
|
||||
sysinfo = "0.32"
|
||||
image = "0.25"
|
||||
oxipng = "9"
|
||||
# default-features = false drops "parallel", which is what actually bounds oxipng's memory:
|
||||
# with rayon it evaluates row filters concurrently, each trial holding its own full-size
|
||||
# buffer, and there is no Options knob to cap that. Without the feature, lib.rs swaps in a
|
||||
# sequential shim (oxipng's own supported path) so peak scales with ONE trial, not N.
|
||||
# PNG optimisation gets slower; it is a background, best-effort, lossless size saving.
|
||||
#
|
||||
# "filetime" must be KEPT: without it OutFile::Path { preserve_attrs: true } silently no-ops.
|
||||
# Dropping "binary" also removes clap/glob/env_logger — a CLI's dependencies that were being
|
||||
# compiled into a server image — and "zopfli", which preset 2 does not use (it selects
|
||||
# Deflaters::Libdeflater, which is not feature-gated).
|
||||
oxipng = { version = "9", default-features = false, features = ["filetime"] }
|
||||
async_zip = { version = "0.0.17", features = ["tokio", "deflate"] }
|
||||
include_dir = "0.7"
|
||||
infer = "0.15"
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
# --- Build stage ---
|
||||
FROM rust:1.87-alpine AS builder
|
||||
FROM rust:1.88-alpine AS builder
|
||||
|
||||
RUN apk add --no-cache musl-dev pkgconfig openssl-dev
|
||||
|
||||
@@ -12,6 +12,7 @@ RUN mkdir src && echo "fn main(){}" > src/main.rs && \
|
||||
|
||||
COPY src ./src
|
||||
COPY static ./static
|
||||
COPY migrations ./migrations
|
||||
RUN touch src/main.rs && cargo build --release
|
||||
|
||||
# --- Runtime stage ---
|
||||
@@ -19,8 +20,18 @@ FROM alpine:3.21
|
||||
|
||||
RUN apk add --no-cache ca-certificates ffmpeg
|
||||
|
||||
# Run as a non-root user. Pre-create and chown the media + export mount paths so
|
||||
# the fresh named volumes inherit the non-root ownership (Docker seeds an empty
|
||||
# named volume from the image directory, preserving its uid/gid) and uploads +
|
||||
# export archives can be written. Exports live OUTSIDE /media on purpose so the
|
||||
# public media ServeDir can't reach them.
|
||||
RUN addgroup -S app && adduser -S app -G app
|
||||
|
||||
WORKDIR /app
|
||||
COPY --from=builder /app/target/release/eventsnap-backend ./
|
||||
|
||||
RUN mkdir -p /media /exports && chown -R app:app /app /media /exports
|
||||
USER app
|
||||
|
||||
EXPOSE 3000
|
||||
CMD ["./eventsnap-backend"]
|
||||
|
||||
2
backend/migrations/008_compression_status.down.sql
Normal file
2
backend/migrations/008_compression_status.down.sql
Normal file
@@ -0,0 +1,2 @@
|
||||
-- Remove compression_status field
|
||||
ALTER TABLE upload DROP COLUMN compression_status;
|
||||
6
backend/migrations/008_compression_status.up.sql
Normal file
6
backend/migrations/008_compression_status.up.sql
Normal file
@@ -0,0 +1,6 @@
|
||||
-- Add compression_status to track media processing state
|
||||
ALTER TABLE upload ADD COLUMN compression_status TEXT NOT NULL DEFAULT 'pending';
|
||||
|
||||
-- Values: 'pending', 'processing', 'done', 'failed'
|
||||
-- Add comment to document the field
|
||||
COMMENT ON COLUMN upload.compression_status IS 'Tracks media compression/preview generation: pending -> processing -> (done or failed)';
|
||||
11
backend/migrations/009_feature_toggles.down.sql
Normal file
11
backend/migrations/009_feature_toggles.down.sql
Normal file
@@ -0,0 +1,11 @@
|
||||
DELETE FROM config WHERE key IN (
|
||||
'rate_limits_enabled',
|
||||
'upload_rate_enabled',
|
||||
'feed_rate_enabled',
|
||||
'export_rate_enabled',
|
||||
'join_rate_enabled',
|
||||
'quota_enabled',
|
||||
'storage_quota_enabled',
|
||||
'upload_count_quota_enabled',
|
||||
'privacy_note'
|
||||
);
|
||||
16
backend/migrations/009_feature_toggles.up.sql
Normal file
16
backend/migrations/009_feature_toggles.up.sql
Normal file
@@ -0,0 +1,16 @@
|
||||
-- Feature toggles for rate limits and quotas, plus the admin-configurable
|
||||
-- Datenschutzhinweis. Everything lives in the `config` table — no schema change.
|
||||
INSERT INTO config (key, value) VALUES
|
||||
-- Rate limits (master + per-endpoint)
|
||||
('rate_limits_enabled', 'true'),
|
||||
('upload_rate_enabled', 'true'),
|
||||
('feed_rate_enabled', 'true'),
|
||||
('export_rate_enabled', 'true'),
|
||||
('join_rate_enabled', 'true'),
|
||||
-- Quotas (master + per-area)
|
||||
('quota_enabled', 'true'),
|
||||
('storage_quota_enabled', 'true'),
|
||||
('upload_count_quota_enabled', 'true'),
|
||||
-- Free-text privacy note shown to guests in My Account. Plain text — no HTML.
|
||||
('privacy_note', '')
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
3
backend/migrations/010_review_indexes.down.sql
Normal file
3
backend/migrations/010_review_indexes.down.sql
Normal file
@@ -0,0 +1,3 @@
|
||||
DROP INDEX IF EXISTS idx_comment_hashtag_hashtag;
|
||||
DROP INDEX IF EXISTS idx_comment_user;
|
||||
DROP INDEX IF EXISTS idx_upload_event_created_id;
|
||||
17
backend/migrations/010_review_indexes.up.sql
Normal file
17
backend/migrations/010_review_indexes.up.sql
Normal file
@@ -0,0 +1,17 @@
|
||||
-- Composite feed index with an id tiebreaker so keyset pagination
|
||||
-- (ORDER BY created_at DESC, id DESC) stays index-covered and stable when
|
||||
-- multiple uploads share a created_at timestamp.
|
||||
CREATE INDEX idx_upload_event_created_id
|
||||
ON upload(event_id, created_at DESC, id DESC)
|
||||
WHERE deleted_at IS NULL;
|
||||
|
||||
-- A user's own comments (moderation, "who commented"). The sibling
|
||||
-- idx_upload_user already exists for uploads; comment(user_id) was missing.
|
||||
CREATE INDEX idx_comment_user
|
||||
ON comment(user_id)
|
||||
WHERE deleted_at IS NULL;
|
||||
|
||||
-- Hashtag filtering over comments — mirrors idx_upload_hashtag_hashtag, which
|
||||
-- only covered upload_hashtag.
|
||||
CREATE INDEX idx_comment_hashtag_hashtag
|
||||
ON comment_hashtag(hashtag_id);
|
||||
25
backend/migrations/011_ban_hides_and_pin_reset.down.sql
Normal file
25
backend/migrations/011_ban_hides_and_pin_reset.down.sql
Normal file
@@ -0,0 +1,25 @@
|
||||
DROP TABLE IF EXISTS pin_reset_request;
|
||||
|
||||
-- Restore v_feed without the is_banned filter.
|
||||
CREATE OR REPLACE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
COUNT(DISTINCT l.user_id) AS like_count,
|
||||
COUNT(DISTINCT c.id) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
LEFT JOIN "like" l ON l.upload_id = u.id
|
||||
LEFT JOIN comment c ON c.upload_id = u.id AND c.deleted_at IS NULL
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
GROUP BY u.id, usr.display_name, usr.is_banned, usr.uploads_hidden;
|
||||
39
backend/migrations/011_ban_hides_and_pin_reset.up.sql
Normal file
39
backend/migrations/011_ban_hides_and_pin_reset.up.sql
Normal file
@@ -0,0 +1,39 @@
|
||||
-- Ban now ALWAYS hides: exclude banned uploaders from the feed view. Previously ban and
|
||||
-- hide were decoupled (a host could ban without hiding), leaving a banned user's photos on
|
||||
-- the feed and baked into the export. `find_visible_media` and the export query get the
|
||||
-- same `is_banned = FALSE` filter in code.
|
||||
CREATE OR REPLACE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
COUNT(DISTINCT l.user_id) AS like_count,
|
||||
COUNT(DISTINCT c.id) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
LEFT JOIN "like" l ON l.upload_id = u.id
|
||||
LEFT JOIN comment c ON c.upload_id = u.id AND c.deleted_at IS NULL
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
AND usr.is_banned = FALSE
|
||||
GROUP BY u.id, usr.display_name, usr.is_banned, usr.uploads_hidden;
|
||||
|
||||
-- pin_reset_request: a guest who forgot their PIN (localStorage was the only copy) can ask
|
||||
-- a host to reset it in-app, instead of being permanently orphaned. One pending request per
|
||||
-- user; cleared when the host resets the PIN or dismisses it, and cascades if the user/event
|
||||
-- is removed.
|
||||
CREATE TABLE pin_reset_request (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
event_id UUID NOT NULL REFERENCES event(id) ON DELETE CASCADE,
|
||||
user_id UUID NOT NULL REFERENCES "user"(id) ON DELETE CASCADE,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
||||
UNIQUE (user_id)
|
||||
);
|
||||
2
backend/migrations/012_export_release_seq.down.sql
Normal file
2
backend/migrations/012_export_release_seq.down.sql
Normal file
@@ -0,0 +1,2 @@
|
||||
ALTER TABLE export_job
|
||||
DROP COLUMN IF EXISTS release_seq;
|
||||
11
backend/migrations/012_export_release_seq.up.sql
Normal file
11
backend/migrations/012_export_release_seq.up.sql
Normal file
@@ -0,0 +1,11 @@
|
||||
-- H1: export generation guard. A reopen (which clears `export_released_at`) followed by a
|
||||
-- re-release could spawn a fresh export worker while a worker from the PRIOR release was
|
||||
-- still streaming a now-stale snapshot to `Gallery.zip`. The stale worker's finalize then
|
||||
-- re-set `export_*_ready = TRUE` on an archive that predated the reopen — a silent,
|
||||
-- permanent data loss (new uploads missing from a "ready" keepsake).
|
||||
--
|
||||
-- `release_seq` is a per-(event,type) generation counter bumped on every (re)release.
|
||||
-- A worker captures the seq it claimed and only finalizes / flips the ready flag while
|
||||
-- that seq is still current; a superseded worker discards its output instead.
|
||||
ALTER TABLE export_job
|
||||
ADD COLUMN release_seq BIGINT NOT NULL DEFAULT 0;
|
||||
2
backend/migrations/013_uploads_hidden_at.down.sql
Normal file
2
backend/migrations/013_uploads_hidden_at.down.sql
Normal file
@@ -0,0 +1,2 @@
|
||||
ALTER TABLE "user"
|
||||
DROP COLUMN IF EXISTS uploads_hidden_at;
|
||||
12
backend/migrations/013_uploads_hidden_at.up.sql
Normal file
12
backend/migrations/013_uploads_hidden_at.up.sql
Normal file
@@ -0,0 +1,12 @@
|
||||
-- Ban-replay on reconnect. A ban is not a soft-delete (it sets `uploads_hidden`/`is_banned`,
|
||||
-- never `upload.deleted_at`), and live eviction rode only on the ephemeral `user-hidden` SSE
|
||||
-- broadcast — which has no reconnect-replay. So a client (especially the unattended diashow
|
||||
-- projector) that missed that broadcast kept cycling the banned user's already-loaded slides.
|
||||
--
|
||||
-- `uploads_hidden_at` stamps WHEN a user's uploads became hidden, so the reconnect delta can
|
||||
-- return the set of users hidden since the client's cursor and the client can evict them.
|
||||
ALTER TABLE "user"
|
||||
ADD COLUMN uploads_hidden_at TIMESTAMPTZ;
|
||||
|
||||
-- Backfill already-hidden users so a reconnect right after this migration still evicts them.
|
||||
UPDATE "user" SET uploads_hidden_at = NOW() WHERE uploads_hidden = TRUE;
|
||||
28
backend/migrations/014_export_epoch.down.sql
Normal file
28
backend/migrations/014_export_epoch.down.sql
Normal file
@@ -0,0 +1,28 @@
|
||||
-- Restore the stored ready flags and the per-job counter.
|
||||
DROP VIEW IF EXISTS export_current;
|
||||
|
||||
ALTER TABLE event ADD COLUMN export_zip_ready BOOLEAN NOT NULL DEFAULT FALSE;
|
||||
ALTER TABLE event ADD COLUMN export_html_ready BOOLEAN NOT NULL DEFAULT FALSE;
|
||||
|
||||
-- Re-derive the flags from what the epoch model considers downloadable, so the old code sees
|
||||
-- exactly the state it would have written itself.
|
||||
UPDATE event e
|
||||
SET export_zip_ready = EXISTS (
|
||||
SELECT 1 FROM export_job j
|
||||
WHERE j.event_id = e.id AND j.type = 'zip'
|
||||
AND j.status = 'done' AND j.epoch = e.export_epoch
|
||||
),
|
||||
export_html_ready = EXISTS (
|
||||
SELECT 1 FROM export_job j
|
||||
WHERE j.event_id = e.id AND j.type = 'html'
|
||||
AND j.status = 'done' AND j.epoch = e.export_epoch
|
||||
)
|
||||
WHERE e.export_released_at IS NOT NULL;
|
||||
|
||||
ALTER TABLE export_job RENAME COLUMN epoch TO release_seq;
|
||||
|
||||
-- The old code treats release_seq as a non-negative per-job counter; the -1 retirement sentinel
|
||||
-- has no meaning there, so floor it back to 0.
|
||||
UPDATE export_job SET release_seq = 0 WHERE release_seq < 0;
|
||||
|
||||
ALTER TABLE event DROP COLUMN export_epoch;
|
||||
86
backend/migrations/014_export_epoch.up.sql
Normal file
86
backend/migrations/014_export_epoch.up.sql
Normal file
@@ -0,0 +1,86 @@
|
||||
-- Export generation, unified.
|
||||
--
|
||||
-- ⚠ ROLLBACK RUNBOOK. This is the repo's first DESTRUCTIVE migration (it DROPs two columns). The
|
||||
-- previous image will NOT boot against this schema: sqlx runs migrations before serving and errors
|
||||
-- with VersionMissing on an unknown version, so you get a crash loop, not a degraded service.
|
||||
-- To roll back to the previous release you must run 014_export_epoch.down.sql BY HAND FIRST, then
|
||||
-- deploy the old image. The down migration re-derives the old ready flags from the epoch state, so
|
||||
-- no keepsake is lost.
|
||||
--
|
||||
-- Before this migration, "which generation of the keepsake is current?" had no single answer.
|
||||
-- It was assembled at runtime from three separately-written pieces of state across two tables:
|
||||
-- * event.export_released_at — a release marker with NO identity (you cannot ask *which* release)
|
||||
-- * export_job.release_seq — a per-row counter, in a different table, bumped in a different statement
|
||||
-- * event.export_{zip,html}_ready — a CACHED COPY of the derivation of the other two
|
||||
-- Keeping those three in agreement took ~10 hand-written guards, and every guard was a repair of the
|
||||
-- same missing invariant. Three consecutive review rounds each found another path that slipped through.
|
||||
--
|
||||
-- This replaces all of it with ONE authority:
|
||||
--
|
||||
-- event.export_epoch — a monotonic counter bumped in the SAME UPDATE as any change to
|
||||
-- export_released_at (release and reopen are its only writers).
|
||||
-- export_job.epoch — a COPY of event.export_epoch taken when the job was enqueued.
|
||||
-- NEVER independently incremented.
|
||||
--
|
||||
-- The single invariant, which replaces the entire proof:
|
||||
--
|
||||
-- An export is downloadable IFF
|
||||
-- event.export_released_at IS NOT NULL
|
||||
-- AND export_job.epoch = event.export_epoch
|
||||
-- AND export_job.status = 'done'
|
||||
--
|
||||
-- Readiness is therefore DERIVED, never stored — so it cannot drift, and no worker can set it.
|
||||
-- A worker holding a dead epoch is INERT BY CONSTRUCTION: anything it writes is invisible to every
|
||||
-- reader, with no guard involved. Losing a race can no longer corrupt state; it can only waste work.
|
||||
--
|
||||
-- Crucially, epoch equality is checked on the SAME ROW the worker updates (export_job), never via a
|
||||
-- cross-table EXISTS. That matters: under READ COMMITTED, when an UPDATE blocks on a row lock and the
|
||||
-- blocker commits, Postgres re-evaluates the WHERE against the *updated target row* but answers
|
||||
-- subqueries on OTHER tables from the statement's ORIGINAL snapshot. The old cross-row guard
|
||||
-- (`EXISTS (SELECT 1 FROM event WHERE ... export_released_at IS NOT NULL)`) was unsound for exactly
|
||||
-- that reason — it could admit a claim against an already-reopened event while RETURNING the
|
||||
-- post-bump seq. A same-row `epoch = $n` predicate is re-evaluated correctly by EPQ.
|
||||
|
||||
ALTER TABLE event ADD COLUMN export_epoch BIGINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- A currently-released event's live generation becomes epoch 1.
|
||||
UPDATE event SET export_epoch = 1 WHERE export_released_at IS NOT NULL;
|
||||
|
||||
-- The per-job counter becomes a plain copy of the event epoch it was enqueued for.
|
||||
ALTER TABLE export_job RENAME COLUMN release_seq TO epoch;
|
||||
|
||||
-- Carry the OLD notion of "downloadable" across exactly: a job the old model considered ready
|
||||
-- (released + done + its ready flag) adopts the event's live epoch. Everything else is retired to
|
||||
-- -1, which can never equal a non-negative event.export_epoch, so it is inert forever.
|
||||
UPDATE export_job j
|
||||
SET epoch = CASE
|
||||
WHEN e.export_released_at IS NOT NULL
|
||||
AND j.status = 'done'
|
||||
AND ((j.type = 'zip' AND e.export_zip_ready)
|
||||
OR (j.type = 'html' AND e.export_html_ready))
|
||||
THEN e.export_epoch
|
||||
ELSE -1
|
||||
END
|
||||
FROM event e
|
||||
WHERE e.id = j.event_id;
|
||||
|
||||
-- The duplicated derivation. Gone — along with the whole class of bugs where a stale worker
|
||||
-- resurrected it, or where it disagreed with the job row it was supposed to summarise.
|
||||
ALTER TABLE event DROP COLUMN export_zip_ready;
|
||||
ALTER TABLE event DROP COLUMN export_html_ready;
|
||||
|
||||
-- The ONE definition of "the current generation". Every reader goes through this, so readiness and
|
||||
-- the file path are resolved in a SINGLE read — closing the download-path TOCTOU where the ready
|
||||
-- flag was checked in one query and file_path fetched in another (a reopen between the two served a
|
||||
-- keepsake that should have 404'd).
|
||||
CREATE VIEW export_current AS
|
||||
SELECT j.event_id,
|
||||
j.type,
|
||||
j.status,
|
||||
j.progress_pct,
|
||||
j.file_path,
|
||||
j.epoch
|
||||
FROM export_job j
|
||||
JOIN event e ON e.id = j.event_id
|
||||
WHERE e.export_released_at IS NOT NULL -- implied by the epoch match; kept as an assertion
|
||||
AND j.epoch = e.export_epoch;
|
||||
3
backend/migrations/015_raise_upload_rate.down.sql
Normal file
3
backend/migrations/015_raise_upload_rate.down.sql
Normal file
@@ -0,0 +1,3 @@
|
||||
-- Revert the default upload rate to 10/hour for installs still on the raised
|
||||
-- default (preserves any explicit admin override at another value).
|
||||
UPDATE config SET value = '10' WHERE key = 'upload_rate_per_hour' AND value = '100';
|
||||
10
backend/migrations/015_raise_upload_rate.up.sql
Normal file
10
backend/migrations/015_raise_upload_rate.up.sql
Normal file
@@ -0,0 +1,10 @@
|
||||
-- Raise the default per-guest upload rate from 10/hour to 100/hour.
|
||||
--
|
||||
-- Rationale: guests routinely upload a burst of 10-20 photos at once (phone
|
||||
-- multi-select). At the old default of 10/hour a real guest's first burst was
|
||||
-- throttled — surfaced by the 2026-07-18 load test. 100/hour comfortably covers
|
||||
-- several bursts across an event while still bounding abuse.
|
||||
--
|
||||
-- Only bump installs still on the old default; an admin who deliberately set a
|
||||
-- different value keeps it (migration 005 seeded 10; this UPDATE is scoped to '10').
|
||||
UPDATE config SET value = '100' WHERE key = 'upload_rate_per_hour' AND value = '10';
|
||||
28
backend/migrations/016_display_derivative.down.sql
Normal file
28
backend/migrations/016_display_derivative.down.sql
Normal file
@@ -0,0 +1,28 @@
|
||||
-- Drop the view (frees the column dependency), remove the column, then restore the
|
||||
-- pre-016 view definition (matches migration 011).
|
||||
DROP VIEW IF EXISTS v_feed;
|
||||
ALTER TABLE upload DROP COLUMN display_path;
|
||||
|
||||
CREATE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
COUNT(DISTINCT l.user_id) AS like_count,
|
||||
COUNT(DISTINCT c.id) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
LEFT JOIN "like" l ON l.upload_id = u.id
|
||||
LEFT JOIN comment c ON c.upload_id = u.id AND c.deleted_at IS NULL
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
AND usr.is_banned = FALSE
|
||||
GROUP BY u.id, usr.display_name, usr.is_banned, usr.uploads_hidden;
|
||||
34
backend/migrations/016_display_derivative.up.sql
Normal file
34
backend/migrations/016_display_derivative.up.sql
Normal file
@@ -0,0 +1,34 @@
|
||||
-- Display derivative: a big-screen-quality image (~2048px long edge) for the diashow.
|
||||
-- The 800px `preview_path` is sized for phone feeds (data saver); upscaled on a projector
|
||||
-- it looks soft. The diashow uses `display_path` instead — bounded in size (safe to decode
|
||||
-- on weak kiosk hardware) yet sharp on 1080p/4K. NULL until the compression worker (or the
|
||||
-- one-time backfill) generates it; consumers fall back to the original when absent.
|
||||
ALTER TABLE upload ADD COLUMN display_path TEXT;
|
||||
|
||||
-- Recreate (not CREATE OR REPLACE, which only allows appending columns at the end) so the
|
||||
-- new column can sit alongside preview_path/thumbnail_path.
|
||||
DROP VIEW IF EXISTS v_feed;
|
||||
CREATE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.display_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
COUNT(DISTINCT l.user_id) AS like_count,
|
||||
COUNT(DISTINCT c.id) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
LEFT JOIN "like" l ON l.upload_id = u.id
|
||||
LEFT JOIN comment c ON c.upload_id = u.id AND c.deleted_at IS NULL
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
AND usr.is_banned = FALSE
|
||||
GROUP BY u.id, usr.display_name, usr.is_banned, usr.uploads_hidden;
|
||||
1
backend/migrations/017_join_ip_rate.down.sql
Normal file
1
backend/migrations/017_join_ip_rate.down.sql
Normal file
@@ -0,0 +1 @@
|
||||
DELETE FROM config WHERE key IN ('join_ip_rate_per_min', 'admin_login_rate_enabled');
|
||||
18
backend/migrations/017_join_ip_rate.up.sql
Normal file
18
backend/migrations/017_join_ip_rate.up.sql
Normal file
@@ -0,0 +1,18 @@
|
||||
-- Per-IP flood ceiling for /join, and the `admin_login_rate_enabled` toggle that
|
||||
-- every prior migration forgot to seed.
|
||||
--
|
||||
-- Rationale: /join was throttled at 5 requests per 60s keyed on the client IP. At a
|
||||
-- venue every guest is behind one NAT, so the whole party shared a single bucket —
|
||||
-- 12 guests scanning the QR code within a few seconds meant 5 got in and 7 were
|
||||
-- turned away. The handler now keys the real anti-spam bucket per (ip, name), the
|
||||
-- same shape as `recover:{ip}:{name}`, and keeps only a loose per-IP ceiling to bound
|
||||
-- raw volume. 60/min comfortably covers a whole wedding arriving at once while still
|
||||
-- capping a flood from a single source.
|
||||
--
|
||||
-- `admin_login_rate_enabled` is read by auth::handlers::admin_login with a code
|
||||
-- default of `true`, but no migration ever inserted it, so it was invisible to the
|
||||
-- admin config UI and to the e2e reseed. Seed it explicitly.
|
||||
INSERT INTO config (key, value) VALUES
|
||||
('join_ip_rate_per_min', '60'),
|
||||
('admin_login_rate_enabled', 'true')
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
2
backend/migrations/018_derivatives_rev.down.sql
Normal file
2
backend/migrations/018_derivatives_rev.down.sql
Normal file
@@ -0,0 +1,2 @@
|
||||
DROP INDEX IF EXISTS idx_upload_derivatives_rev;
|
||||
ALTER TABLE upload DROP COLUMN IF EXISTS derivatives_rev;
|
||||
18
backend/migrations/018_derivatives_rev.up.sql
Normal file
18
backend/migrations/018_derivatives_rev.up.sql
Normal file
@@ -0,0 +1,18 @@
|
||||
-- Track which revision of the derivative pipeline produced an upload's preview/display.
|
||||
--
|
||||
-- Rev 1 applies the EXIF orientation tag. Everything generated before it decoded the raw
|
||||
-- sensor pixels and re-encoded to JPEG (which writes no EXIF), so every portrait phone photo
|
||||
-- was stored sideways in the feed preview, the diashow display and the keepsake — while the
|
||||
-- untouched original still rendered upright.
|
||||
--
|
||||
-- Existing rows default to 0 so the startup backfill can find and re-generate them exactly
|
||||
-- once; bump the constant in services/compression.rs if the pipeline ever changes again.
|
||||
ALTER TABLE upload ADD COLUMN IF NOT EXISTS derivatives_rev SMALLINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- Only image derivatives are affected — video thumbnails are extracted by ffmpeg, which
|
||||
-- already honours the rotation matrix. Mark them current so the backfill skips them.
|
||||
UPDATE upload SET derivatives_rev = 1 WHERE mime_type NOT LIKE 'image/%';
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_upload_derivatives_rev
|
||||
ON upload (derivatives_rev)
|
||||
WHERE deleted_at IS NULL;
|
||||
1
backend/migrations/019_recover_ip_rate.down.sql
Normal file
1
backend/migrations/019_recover_ip_rate.down.sql
Normal file
@@ -0,0 +1 @@
|
||||
DELETE FROM config WHERE key = 'recover_ip_rate_per_min';
|
||||
19
backend/migrations/019_recover_ip_rate.up.sql
Normal file
19
backend/migrations/019_recover_ip_rate.up.sql
Normal file
@@ -0,0 +1,19 @@
|
||||
-- Per-IP flood ceiling for /recover, mirroring the one migration 017 added for /join.
|
||||
--
|
||||
-- Rationale: /recover is keyed `recover:{ip}:{name}` at 5 per 15 minutes. That is the
|
||||
-- right shape for its actual job — stopping someone who knows a display name (they are
|
||||
-- visible on the feed) from burning the victim's 3-strike PIN counter and locking them
|
||||
-- out repeatedly. But the name is attacker-chosen, so cycling names mints a fresh bucket
|
||||
-- every time and the per-IP cost is unbounded.
|
||||
--
|
||||
-- Behind that limiter sits a cost-12 bcrypt verify, including an UNCONDITIONAL throwaway
|
||||
-- verify for names that don't exist — deliberately, to close a timing oracle. So an
|
||||
-- unknown name is the cheapest possible way to make the server do ~200ms of hashing.
|
||||
-- Without a ceiling, one client can saturate the box's CPU with a name generator.
|
||||
--
|
||||
-- 30/min is far above any real recovery attempt (a guest tries their PIN a handful of
|
||||
-- times) while capping a name-cycling flood. The per-(ip, name) bucket is unchanged and
|
||||
-- remains the anti-guessing control.
|
||||
INSERT INTO config (key, value) VALUES
|
||||
('recover_ip_rate_per_min', '30')
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
1
backend/migrations/020_social_rate.down.sql
Normal file
1
backend/migrations/020_social_rate.down.sql
Normal file
@@ -0,0 +1 @@
|
||||
DELETE FROM config WHERE key IN ('social_rate_per_min', 'social_rate_enabled');
|
||||
16
backend/migrations/020_social_rate.up.sql
Normal file
16
backend/migrations/020_social_rate.up.sql
Normal file
@@ -0,0 +1,16 @@
|
||||
-- Per-user rate limit for social writes (likes, comments, comment deletions).
|
||||
--
|
||||
-- These were the only writes in the app with no limit at all. Every other mutating
|
||||
-- path -- upload, join, recover, export, admin login -- carries one; social.rs
|
||||
-- carried none, so the coverage was asymmetric rather than deliberately open.
|
||||
--
|
||||
-- Severity is genuinely low for an invited-guest event, and the amplification worry
|
||||
-- turned out to be contained: a like fans an SSE broadcast to ~100 clients, but the
|
||||
-- export regeneration it could otherwise trigger is debounced (REGEN_DEBOUNCE 20s)
|
||||
-- and superseded workers are inert. So this closes the gap for symmetry, not urgency,
|
||||
-- and the ceiling is set high enough that no real guest will ever meet it -- a
|
||||
-- double-tapping enthusiast at a wedding is not the thing being defended against.
|
||||
INSERT INTO config (key, value) VALUES
|
||||
('social_rate_per_min', '120'),
|
||||
('social_rate_enabled', 'true')
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
3
backend/migrations/021_derivative_attempts.down.sql
Normal file
3
backend/migrations/021_derivative_attempts.down.sql
Normal file
@@ -0,0 +1,3 @@
|
||||
DROP INDEX IF EXISTS idx_upload_derivative_backfill;
|
||||
ALTER TABLE upload DROP COLUMN IF EXISTS derivative_last_error;
|
||||
ALTER TABLE upload DROP COLUMN IF EXISTS derivative_attempts;
|
||||
23
backend/migrations/021_derivative_attempts.up.sql
Normal file
23
backend/migrations/021_derivative_attempts.up.sql
Normal file
@@ -0,0 +1,23 @@
|
||||
-- Bound how many times a permanently-failing upload can be re-processed.
|
||||
--
|
||||
-- Without this, one poisoned row is an outage. The upload row is committed BEFORE compression
|
||||
-- starts, `derivatives_rev` defaults to 0, and `set_derivatives_rev` only runs on success — so
|
||||
-- a row whose processing kills the container survives at rev 0, the unconditional startup
|
||||
-- backfill re-selects it on the next boot, and `restart: unless-stopped` turns that into an
|
||||
-- infinite kill loop. Every restart also drops every SSE stream and truncates every in-flight
|
||||
-- upload. That was reachable via a single large PNG (see services/compression.rs), but the
|
||||
-- shape is general: any input that can kill or hang the worker repeats forever.
|
||||
--
|
||||
-- The counter is incremented WRITE-AHEAD, before the work is attempted, because the failure
|
||||
-- mode being defended against is a SIGKILL — no error is returned, no handler runs, no Drop
|
||||
-- fires. A counter bumped in an error path increments zero times per crash and changes nothing.
|
||||
ALTER TABLE upload ADD COLUMN IF NOT EXISTS derivative_attempts SMALLINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- Last failure text, so a row that has given up can be diagnosed without reproducing it.
|
||||
-- Nothing reads this in code; it exists for the operator.
|
||||
ALTER TABLE upload ADD COLUMN IF NOT EXISTS derivative_last_error TEXT;
|
||||
|
||||
-- Serves the backfill selection, which now filters on both columns.
|
||||
CREATE INDEX IF NOT EXISTS idx_upload_derivative_backfill
|
||||
ON upload (derivatives_rev, derivative_attempts)
|
||||
WHERE deleted_at IS NULL;
|
||||
26
backend/migrations/022_feed_scalar_counts.down.sql
Normal file
26
backend/migrations/022_feed_scalar_counts.down.sql
Normal file
@@ -0,0 +1,26 @@
|
||||
-- Restore the 016 definition verbatim.
|
||||
DROP VIEW IF EXISTS v_feed;
|
||||
CREATE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.display_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
COUNT(DISTINCT l.user_id) AS like_count,
|
||||
COUNT(DISTINCT c.id) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
LEFT JOIN "like" l ON l.upload_id = u.id
|
||||
LEFT JOIN comment c ON c.upload_id = u.id AND c.deleted_at IS NULL
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
AND usr.is_banned = FALSE
|
||||
GROUP BY u.id, usr.display_name, usr.is_banned, usr.uploads_hidden;
|
||||
55
backend/migrations/022_feed_scalar_counts.up.sql
Normal file
55
backend/migrations/022_feed_scalar_counts.up.sql
Normal file
@@ -0,0 +1,55 @@
|
||||
-- Make a feed page cost a page, not the whole event.
|
||||
--
|
||||
-- The previous definition (016) computed like_count/comment_count with LEFT JOINs and a
|
||||
-- GROUP BY. Postgres CAN push the `event_id = $1` qual and the keyset predicate through the
|
||||
-- view — verified with EXPLAIN, it uses idx_upload_event_created_id — but it CANNOT push
|
||||
-- ORDER BY ... LIMIT across a GroupAggregate. So every feed request aggregated every upload in
|
||||
-- the event (times its likes and comments) and only then sorted and took 21 rows. The cost
|
||||
-- grew with the event, not with the page, and page 1 — the most expensive one — is exactly
|
||||
-- what refreshFeedInPlace refetches on every completed upload, from every open feed in the
|
||||
-- venue.
|
||||
--
|
||||
-- Correlated scalar subqueries move the counts ABOVE the Limit in the plan: they are evaluated
|
||||
-- once per returned row, so 21 index lookups instead of a full aggregation.
|
||||
--
|
||||
-- The rewrite is EXACTLY equivalent, not merely close:
|
||||
-- * "like" is keyed (upload_id, user_id), so COUNT(DISTINCT l.user_id) == count(*).
|
||||
-- * comment.id is the primary key, so COUNT(DISTINCT c.id) == count(*).
|
||||
-- * one row per upload either way — the GROUP BY was on u.id.
|
||||
-- Column names, order and types are unchanged (count(*) and COUNT(DISTINCT ...) are both
|
||||
-- bigint), so no Rust code changes.
|
||||
--
|
||||
-- No new index needed: idx_like_upload plus the (upload_id, user_id) PK serve the like
|
||||
-- subquery, and idx_comment_upload ... WHERE deleted_at IS NULL matches the comment
|
||||
-- subquery's predicate exactly.
|
||||
--
|
||||
-- One thing a future editor needs to know: the hashtag-filtered feed joins upload_hashtag
|
||||
-- against this view. That was safe before only because the GROUP BY collapsed the join
|
||||
-- fan-out; it is safe now because the view is one row per upload and upload_hashtag is keyed
|
||||
-- (upload_id, hashtag_id) with a single tag filtered. Adding a second tag filter would need
|
||||
-- fresh thought.
|
||||
|
||||
-- Not CASCADE: if something ever comes to depend on this view, the migration should fail
|
||||
-- loudly rather than silently drop it.
|
||||
DROP VIEW IF EXISTS v_feed;
|
||||
CREATE VIEW v_feed AS
|
||||
SELECT
|
||||
u.id,
|
||||
u.event_id,
|
||||
u.user_id,
|
||||
usr.display_name AS uploader_name,
|
||||
usr.is_banned,
|
||||
usr.uploads_hidden,
|
||||
u.preview_path,
|
||||
u.thumbnail_path,
|
||||
u.display_path,
|
||||
u.mime_type,
|
||||
u.caption,
|
||||
u.created_at,
|
||||
(SELECT count(*) FROM "like" l WHERE l.upload_id = u.id) AS like_count,
|
||||
(SELECT count(*) FROM comment c WHERE c.upload_id = u.id AND c.deleted_at IS NULL) AS comment_count
|
||||
FROM upload u
|
||||
JOIN "user" usr ON u.user_id = usr.id
|
||||
WHERE u.deleted_at IS NULL
|
||||
AND usr.uploads_hidden = FALSE
|
||||
AND usr.is_banned = FALSE;
|
||||
11
backend/migrations/023_reserved_names_and_pin_decay.down.sql
Normal file
11
backend/migrations/023_reserved_names_and_pin_decay.down.sql
Normal file
@@ -0,0 +1,11 @@
|
||||
-- NOTE: the reserved-name rename in the up migration is NOT reversible. The original names
|
||||
-- are not recorded anywhere, and reversing it would in any case re-create the state that
|
||||
-- bricked admin login. Rolling back the schema does not roll back that data change.
|
||||
DELETE FROM config WHERE key IN (
|
||||
'recover_name_rate_per_15min',
|
||||
'pin_reset_ip_rate_per_min',
|
||||
'upload_edit_rate_per_min',
|
||||
'upload_edit_rate_enabled'
|
||||
);
|
||||
|
||||
ALTER TABLE "user" DROP COLUMN IF EXISTS last_failed_pin_at;
|
||||
48
backend/migrations/023_reserved_names_and_pin_decay.up.sql
Normal file
48
backend/migrations/023_reserved_names_and_pin_decay.up.sql
Normal file
@@ -0,0 +1,48 @@
|
||||
-- Two independent auth defects that share a migration because they share a table.
|
||||
|
||||
-- 1. RESERVED NAMES — free any guest squatting on a name the admin path used to depend on.
|
||||
--
|
||||
-- Migration 007 made display_name unique per event case-insensitively, and `join` had no
|
||||
-- reserved-name guard. So any guest could join as "admin"/"Admin"/"ADMIN" before the operator's
|
||||
-- first admin login; admin_login then looked its user up BY NAME, missed (wrong role), fell
|
||||
-- through to creating "Admin", violated that unique index, and returned a 500 — permanently,
|
||||
-- with no in-app recovery. Moderation, config and gallery release all gone, fixed only by SQL.
|
||||
--
|
||||
-- The real fix is in code (look the admin up by role, never by name — see auth/handlers.rs).
|
||||
-- This clears the state an already-deployed database may be carrying.
|
||||
--
|
||||
-- RENAMED, NEVER DELETED: the guest keeps their uploads, their PIN and their session. Only
|
||||
-- non-admin rows are touched — a real admin row named "Admin" is the expected state.
|
||||
UPDATE "user" u
|
||||
SET display_name = u.display_name || ' (' || left(u.id::text, 8) || ')'
|
||||
WHERE u.role <> 'admin'
|
||||
AND lower(u.display_name) IN ('admin', 'administrator', 'host', 'eventsnap');
|
||||
|
||||
-- 2. PIN LOCKOUT DECAY.
|
||||
--
|
||||
-- failed_pin_attempts only ever cleared on a successful recovery or after a lockout expired, so
|
||||
-- honest typos accumulated across days: a guest who fat-fingered their PIN twice last night
|
||||
-- arrives today already two-thirds of the way to being locked out. With the threshold now
|
||||
-- raised (see below) a decay window is what keeps that raise safe rather than merely lenient.
|
||||
ALTER TABLE "user" ADD COLUMN IF NOT EXISTS last_failed_pin_at TIMESTAMPTZ;
|
||||
|
||||
-- Rate-limit knobs introduced with this release.
|
||||
--
|
||||
-- recover_name_rate_per_15min (4, was a hardcoded 5): the per-(IP, name) ceiling. It MUST stay
|
||||
-- below the account-lock threshold, which is the whole defect — at 5-per-IP against a 3-strike
|
||||
-- lock, three requests from one IP locked any guest whose name is visible on the feed, every 15
|
||||
-- minutes, forever. The lock threshold moves to 12 in code, so locking a victim now needs at
|
||||
-- least three distinct sources while an honest guest never comes close.
|
||||
--
|
||||
-- pin_reset_ip_rate_per_min (30): /recover/request was the one unauthenticated endpoint with no
|
||||
-- per-IP ceiling at all — /join got one in 017 and /recover in 019, and this third one was
|
||||
-- simply missed. Its per-name key is attacker-chosen, so cycling names minted a fresh bucket
|
||||
-- every time and the per-IP cost was unbounded.
|
||||
--
|
||||
-- upload_edit_rate_per_min (30): PATCH /upload/{id} had no rate limit of any kind.
|
||||
INSERT INTO config (key, value) VALUES
|
||||
('recover_name_rate_per_15min', '4'),
|
||||
('pin_reset_ip_rate_per_min', '30'),
|
||||
('upload_edit_rate_per_min', '30'),
|
||||
('upload_edit_rate_enabled', 'true')
|
||||
ON CONFLICT (key) DO NOTHING;
|
||||
28
backend/migrations/README.md
Normal file
28
backend/migrations/README.md
Normal file
@@ -0,0 +1,28 @@
|
||||
# Migrations
|
||||
|
||||
SQLx-managed Postgres migrations. Each `NNN_topic.up.sql` has a matching
|
||||
`NNN_topic.down.sql`. Run by `sqlx::migrate!()` at app start.
|
||||
|
||||
## Rules
|
||||
|
||||
1. **Never edit a shipped migration.** If a column needs to change or a fix needs to
|
||||
land, write a new migration. Production has already applied the old one and SQLx
|
||||
tracks each by checksum — editing in place will fail to apply on existing databases.
|
||||
2. **Always pair `.up.sql` with a `.down.sql`.** Reverts may not be perfect (data
|
||||
loss is sometimes unavoidable) but the file must exist and do the best it can.
|
||||
3. **Prefer additive changes.** New columns, new tables, new keys in `config`. Drop /
|
||||
rename only when there is no alternative.
|
||||
4. **No business logic in migrations.** Schema + seeds only. Anything that needs Rust
|
||||
code goes in a one-off binary, not a migration file.
|
||||
5. **One concern per migration.** Easier to revert. Easier to read in `git log`.
|
||||
|
||||
## Numbering
|
||||
|
||||
Zero-padded three digits, monotonically increasing. The next free number lives at the
|
||||
bottom of the directory listing — pick that.
|
||||
|
||||
## Seed-only migrations
|
||||
|
||||
When you only need to add `config` keys (feature flags, defaults), use
|
||||
`INSERT … ON CONFLICT DO NOTHING` so existing operator overrides survive. See
|
||||
`009_feature_toggles.up.sql` for the canonical shape.
|
||||
195
backend/scripts/rehearse-014.sh
Executable file
195
backend/scripts/rehearse-014.sh
Executable file
@@ -0,0 +1,195 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Dress rehearsal for migration 014 (export_epoch) — the repo's first DESTRUCTIVE migration.
|
||||
#
|
||||
# 014 DROPs event.export_zip_ready / export_html_ready and rewrites export_job.release_seq into an
|
||||
# epoch. The dangerous part is not the DDL, it's the BACKFILL: it has to carry the old notion of
|
||||
# "downloadable" across exactly. Get it wrong and a released event's keepsake silently 404s for
|
||||
# every guest, on a dataset you cannot re-create.
|
||||
#
|
||||
# This script proves the backfill preserves downloadability, against a scratch copy of a REAL dump:
|
||||
#
|
||||
# ./rehearse-014.sh /path/to/prod-dump.sql # rehearse against production data
|
||||
# ./rehearse-014.sh # synthetic: build a pre-014 DB covering every case
|
||||
#
|
||||
# It asserts:
|
||||
# 1. up: {(event,type) downloadable BEFORE} == {(event,type) downloadable AFTER}
|
||||
# 2. down: the old ready flags come back identical (so a rollback is actually a rollback)
|
||||
# 3. up again: still identical (so a roll-forward after a rollback is safe)
|
||||
#
|
||||
# It NEVER touches your real database. Everything runs in a throwaway container.
|
||||
#
|
||||
# NOTE ON ROLLBACK (see the runbook header in 014_export_epoch.up.sql): sqlx runs migrations before
|
||||
# serving and errors with VersionMissing on an unknown version, so the PREVIOUS image will not boot
|
||||
# against the 014 schema — it crash-loops. Rolling back means running 014_export_epoch.down.sql BY
|
||||
# HAND FIRST, then deploying the old image. Assertion 2 is what makes that safe. Rehearse it before
|
||||
# you need it, not while the party is happening.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
DUMP="${1:-}"
|
||||
MIG_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/../migrations" && pwd)"
|
||||
CONTAINER="eventsnap-rehearse-014"
|
||||
PGPASSWORD="rehearse"
|
||||
DB="eventsnap_rehearsal"
|
||||
|
||||
cleanup() { docker rm -f "$CONTAINER" > /dev/null 2>&1 || true; }
|
||||
trap cleanup EXIT
|
||||
cleanup
|
||||
|
||||
echo "▸ Starting a throwaway postgres:16 (nothing here touches your real DB)…"
|
||||
docker run --rm -d --name "$CONTAINER" \
|
||||
-e POSTGRES_PASSWORD="$PGPASSWORD" -e POSTGRES_DB="$DB" \
|
||||
postgres:16-alpine > /dev/null
|
||||
|
||||
psql() { docker exec -i -e PGPASSWORD="$PGPASSWORD" "$CONTAINER" psql -U postgres -d "$DB" -v ON_ERROR_STOP=1 "$@"; }
|
||||
|
||||
for _ in $(seq 1 30); do
|
||||
if docker exec "$CONTAINER" pg_isready -U postgres > /dev/null 2>&1; then break; fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
if [[ -n "$DUMP" ]]; then
|
||||
echo "▸ Restoring $DUMP …"
|
||||
psql -q < "$DUMP"
|
||||
else
|
||||
echo "▸ No dump given — building a synthetic pre-014 DB (migrations 001→013)…"
|
||||
for f in "$MIG_DIR"/0[0-1][0-9]_*.up.sql; do
|
||||
[[ "$(basename "$f")" == 014_* ]] && continue
|
||||
psql -q < "$f"
|
||||
done
|
||||
|
||||
# Every state the backfill has to classify. If 014 mishandles ANY of these, assertion 1 fails.
|
||||
# Every state the backfill has to classify. export_job is UNIQUE (event_id, type), so each event
|
||||
# has at most one job per type — the cases are distinguished by event, not by piling up rows.
|
||||
# A: released, both flags, both done → both stay downloadable
|
||||
# B: released, ZIP flag only → ZIP downloadable, HTML not (a half-built keepsake)
|
||||
# C: released, jobs done, NO flags → NOT downloadable (a superseded worker's finished job:
|
||||
# the exact state the old ready-flag model used to leak)
|
||||
# D: never released, jobs done → NOT downloadable (reopened, or built before release)
|
||||
# E: released, ZIP flag set but the job is FAILED/RUNNING → NOT downloadable (flag/job disagree,
|
||||
# which the old two-source model made representable at all)
|
||||
echo "▸ Seeding: released+both, zip-only, done-but-stale, unreleased, flag/job-disagreement…"
|
||||
psql -q <<'SQL'
|
||||
INSERT INTO event (id, slug, name, export_released_at, export_zip_ready, export_html_ready) VALUES
|
||||
('aaaaaaaa-0000-0000-0000-000000000001','ev-a','A', now(), TRUE, TRUE),
|
||||
('aaaaaaaa-0000-0000-0000-000000000002','ev-b','B', now(), TRUE, FALSE),
|
||||
('aaaaaaaa-0000-0000-0000-000000000003','ev-c','C', now(), FALSE, FALSE),
|
||||
('aaaaaaaa-0000-0000-0000-000000000004','ev-d','D', NULL, FALSE, FALSE),
|
||||
('aaaaaaaa-0000-0000-0000-000000000005','ev-e','E', now(), TRUE, TRUE);
|
||||
|
||||
INSERT INTO export_job (event_id, type, status, progress_pct, file_path, release_seq)
|
||||
SELECT e.id, t::export_type, 'done'::export_status, 100, '/x/' || e.slug || '.' || t, 0
|
||||
FROM event e CROSS JOIN (VALUES ('zip'),('html')) AS v(t);
|
||||
|
||||
-- E: the flags claim ready, the jobs say otherwise. Old model required BOTH, so neither is
|
||||
-- downloadable — the migration must not "trust the flag" and resurrect them.
|
||||
UPDATE export_job SET status = 'failed', file_path = NULL, progress_pct = 40
|
||||
WHERE event_id = 'aaaaaaaa-0000-0000-0000-000000000005' AND type = 'zip';
|
||||
UPDATE export_job SET status = 'running', file_path = NULL, progress_pct = 70
|
||||
WHERE event_id = 'aaaaaaaa-0000-0000-0000-000000000005' AND type = 'html';
|
||||
SQL
|
||||
fi
|
||||
|
||||
echo "▸ Snapshotting what is downloadable under the OLD model…"
|
||||
psql -q <<'SQL'
|
||||
CREATE TABLE _rehearsal_before AS
|
||||
SELECT j.event_id, j.type
|
||||
FROM export_job j JOIN event e ON e.id = j.event_id
|
||||
WHERE e.export_released_at IS NOT NULL
|
||||
AND j.status = 'done'
|
||||
AND ((j.type = 'zip' AND e.export_zip_ready) OR (j.type = 'html' AND e.export_html_ready));
|
||||
|
||||
-- For assertion 2: the exact flags a rollback has to reproduce.
|
||||
CREATE TABLE _rehearsal_flags AS
|
||||
SELECT id, export_zip_ready, export_html_ready FROM event;
|
||||
SQL
|
||||
|
||||
before=$(psql -tAc "SELECT count(*) FROM _rehearsal_before")
|
||||
echo " → $before (event,type) pair(s) downloadable before."
|
||||
|
||||
# The assertion. A symmetric difference of zero means the backfill carried the old notion of
|
||||
# "downloadable" across EXACTLY: nothing gained (a stale keepsake resurrected), nothing lost (a
|
||||
# guest's keepsake silently 404s).
|
||||
assert_up_preserves() {
|
||||
local label="$1" diff
|
||||
diff=$(psql -tAc "
|
||||
SELECT count(*) FROM (
|
||||
(SELECT event_id, type FROM _rehearsal_before
|
||||
EXCEPT SELECT event_id, type FROM export_current WHERE status = 'done')
|
||||
UNION ALL
|
||||
(SELECT event_id, type FROM export_current WHERE status = 'done'
|
||||
EXCEPT SELECT event_id, type FROM _rehearsal_before)
|
||||
) d")
|
||||
if [[ "$diff" != "0" ]]; then
|
||||
echo "✗ FAIL ($label): $diff (event,type) pair(s) changed downloadability across the migration."
|
||||
psql -c "
|
||||
(SELECT 'LOST — guests can no longer download this' AS problem, event_id, type FROM _rehearsal_before
|
||||
EXCEPT ALL SELECT 'LOST — guests can no longer download this', event_id, type FROM export_current WHERE status='done')
|
||||
UNION ALL
|
||||
(SELECT 'GAINED — a stale keepsake became downloadable', event_id, type FROM export_current WHERE status='done'
|
||||
EXCEPT ALL SELECT 'GAINED — a stale keepsake became downloadable', event_id, type FROM _rehearsal_before)"
|
||||
exit 1
|
||||
fi
|
||||
echo "✓ $label: downloadability preserved exactly ($before pair(s))."
|
||||
}
|
||||
|
||||
echo "▸ Applying 014 (up)…"
|
||||
psql -q < "$MIG_DIR/014_export_epoch.up.sql"
|
||||
assert_up_preserves "up"
|
||||
|
||||
echo "▸ Applying 014 (down) — the rollback path…"
|
||||
psql -q < "$MIG_DIR/014_export_epoch.down.sql"
|
||||
|
||||
# The down migration is an inverse UP TO DRIFT, not a bit-exact inverse — deliberately.
|
||||
#
|
||||
# The old model let a ready flag disagree with its job row (flag TRUE over a running/failed job):
|
||||
# the flag was a CACHED copy of a derivation, and keeping a cache in agreement with its source
|
||||
# across concurrent workers is precisely what it kept failing at. The down migration re-derives the
|
||||
# flags from the epoch state, so such a pair comes back FALSE. That HEALS the drift rather than
|
||||
# faithfully restoring corruption, and it costs nothing: a flag over a non-done job had no file to
|
||||
# serve (file_path is NULL until the worker finishes), so the old code 404'd on it anyway.
|
||||
#
|
||||
# What a rollback must NOT do is drop a flag for a keepsake that was genuinely downloadable
|
||||
# (flag set AND the job actually done). That is the assertion.
|
||||
lost=$(psql -tAc "
|
||||
SELECT count(*) FROM _rehearsal_before b
|
||||
WHERE NOT EXISTS (
|
||||
SELECT 1 FROM event e
|
||||
WHERE e.id = b.event_id
|
||||
AND ((b.type = 'zip' AND e.export_zip_ready) OR (b.type = 'html' AND e.export_html_ready)))")
|
||||
if [[ "$lost" != "0" ]]; then
|
||||
echo "✗ FAIL (down): $lost downloadable keepsake(s) lost their ready flag — the rollback is LOSSY."
|
||||
psql -c "
|
||||
SELECT b.event_id, b.type, 'was downloadable, flag not restored' AS problem
|
||||
FROM _rehearsal_before b
|
||||
WHERE NOT EXISTS (
|
||||
SELECT 1 FROM event e
|
||||
WHERE e.id = b.event_id
|
||||
AND ((b.type = 'zip' AND e.export_zip_ready) OR (b.type = 'html' AND e.export_html_ready)))"
|
||||
exit 1
|
||||
fi
|
||||
echo "✓ down: every downloadable keepsake kept its flag — the old image sees a working state."
|
||||
|
||||
healed=$(psql -tAc "
|
||||
SELECT count(*) FROM event e JOIN _rehearsal_flags f ON f.id = e.id
|
||||
WHERE (e.export_zip_ready, e.export_html_ready) IS DISTINCT FROM (f.export_zip_ready, f.export_html_ready)")
|
||||
if [[ "$healed" != "0" ]]; then
|
||||
echo " ℹ $healed event(s) came back with a CLEARED flag that had drifted from its job row."
|
||||
echo " Not a loss — those had no file to serve. The rollback normalises them; the old code"
|
||||
echo " will rebuild on the next release. Listing them so the behaviour is not a surprise:"
|
||||
psql -c "SELECT e.id, f.export_zip_ready AS was_zip, e.export_zip_ready AS now_zip,
|
||||
f.export_html_ready AS was_html, e.export_html_ready AS now_html
|
||||
FROM event e JOIN _rehearsal_flags f ON f.id = e.id
|
||||
WHERE (e.export_zip_ready, e.export_html_ready)
|
||||
IS DISTINCT FROM (f.export_zip_ready, f.export_html_ready)"
|
||||
fi
|
||||
|
||||
echo "▸ Re-applying 014 (up) — roll forward after a rollback…"
|
||||
psql -q < "$MIG_DIR/014_export_epoch.up.sql"
|
||||
assert_up_preserves "up (after rollback)"
|
||||
|
||||
echo
|
||||
echo "✓ Rehearsal passed. 014 is safe to deploy against this dataset."
|
||||
[[ -z "$DUMP" ]] && echo " (synthetic data — re-run with a real pg_dump before you deploy for real)"
|
||||
exit 0
|
||||
@@ -1,11 +1,12 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use axum::extract::State;
|
||||
use axum::http::{HeaderMap, StatusCode};
|
||||
use axum::Json;
|
||||
use axum::extract::{ConnectInfo, State};
|
||||
use axum::http::{HeaderMap, StatusCode};
|
||||
use chrono::Utc;
|
||||
use rand::Rng;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::net::SocketAddr;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::auth::jwt;
|
||||
@@ -14,9 +15,49 @@ use crate::error::AppError;
|
||||
use crate::models::event::Event;
|
||||
use crate::models::session::Session;
|
||||
use crate::models::user::{User, UserRole};
|
||||
use crate::services::config;
|
||||
use crate::services::rate_limiter::client_ip;
|
||||
use crate::state::AppState;
|
||||
|
||||
/// Names a guest may not take.
|
||||
///
|
||||
/// Defence in depth only. The real fix for the admin-lockout defect is that `admin_login` now
|
||||
/// resolves its user by ROLE rather than by name (see `User::find_admin_for_event`), which is
|
||||
/// why homoglyph and zero-width bypasses of this list are not a concern: the name is no longer
|
||||
/// load-bearing for anything. What this buys is that a guest cannot impersonate the host in the
|
||||
/// feed's byline, and that "Admin" stays available for the admin row.
|
||||
const RESERVED_DISPLAY_NAMES: &[&str] = &["admin", "administrator", "host", "eventsnap"];
|
||||
|
||||
fn is_reserved_display_name(name: &str) -> bool {
|
||||
let name = name.trim().to_lowercase();
|
||||
RESERVED_DISPLAY_NAMES.contains(&name.as_str())
|
||||
}
|
||||
|
||||
/// Trim and bounds-check a display name.
|
||||
///
|
||||
/// Shared by `join`, `recover` and `request_pin_reset` so the length check happens BEFORE the
|
||||
/// name is used to build a rate-limiter key. It was inline in `join` only, so on the other two
|
||||
/// endpoints `format!("...:{ip}:{name_key}")` allocated from an unbounded, attacker-chosen
|
||||
/// string and stored it in a HashMap pruned once an hour with a 24 h ceiling — turning the
|
||||
/// limiter itself into the memory-exhaustion primitive it exists to prevent.
|
||||
fn validate_display_name(raw: &str) -> Result<&str, AppError> {
|
||||
let name = raw.trim();
|
||||
let chars = name.chars().count();
|
||||
if chars == 0 || chars > 50 {
|
||||
return Err(AppError::BadRequest(
|
||||
"Name muss zwischen 1 und 50 Zeichen lang sein.".into(),
|
||||
));
|
||||
}
|
||||
// Postgres rejects 0x00 in TEXT columns with a 500. Catch it here so callers see a clean
|
||||
// 400 instead of an internal error.
|
||||
if name.contains('\0') {
|
||||
return Err(AppError::BadRequest(
|
||||
"Name enthält ungültige Zeichen.".into(),
|
||||
));
|
||||
}
|
||||
Ok(name)
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct JoinRequest {
|
||||
pub display_name: String,
|
||||
@@ -32,22 +73,58 @@ pub struct JoinResponse {
|
||||
|
||||
pub async fn join(
|
||||
State(state): State<AppState>,
|
||||
ConnectInfo(peer): ConnectInfo<SocketAddr>,
|
||||
headers: HeaderMap,
|
||||
Json(body): Json<JoinRequest>,
|
||||
) -> Result<(StatusCode, Json<JoinResponse>), AppError> {
|
||||
let ip = client_ip(&headers, "unknown");
|
||||
if !state.rate_limiter.check(format!("join:{ip}"), 5, Duration::from_secs(60)) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
None,
|
||||
));
|
||||
let ip = client_ip(&headers, &peer.ip().to_string());
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let join_rate_on = config::get_bool(&state.config_cache, "join_rate_enabled", true).await;
|
||||
|
||||
// Coarse per-IP flood ceiling. `/join` is pre-auth so there is no user to key on, and
|
||||
// at a venue EVERY guest arrives from one public IP — a tight per-IP bucket meant the
|
||||
// 6th person through the door was turned away by the 5 ahead of them. So the per-IP
|
||||
// limit here only bounds raw volume; the real anti-spam bucket is per-name below.
|
||||
// Cheap enough to run before validation, which keeps a flood of malformed bodies from
|
||||
// being free.
|
||||
if rate_limits_on && join_rate_on {
|
||||
let ip_ceiling = config::get_usize(&state.config_cache, "join_ip_rate_per_min", 60).await;
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("join_ip:{ip}"),
|
||||
ip_ceiling,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let display_name = body.display_name.trim();
|
||||
if display_name.is_empty() || display_name.len() > 50 {
|
||||
return Err(AppError::BadRequest(
|
||||
"Name muss zwischen 1 und 50 Zeichen lang sein.".into(),
|
||||
));
|
||||
let display_name = validate_display_name(&body.display_name)?;
|
||||
if is_reserved_display_name(display_name) {
|
||||
// 409, matching the name-taken response below, so the frontend's existing handling
|
||||
// works unchanged. See RESERVED_DISPLAY_NAMES for why this exists.
|
||||
return Err(AppError::Conflict(format!(
|
||||
"Der Name \"{display_name}\" ist reserviert. Bitte wähle einen anderen."
|
||||
)));
|
||||
}
|
||||
|
||||
// Per-guest bucket, keyed like the `recover:{ip}:{name}` limiter below. This carries
|
||||
// the original 5/60s anti-spam intent, but one guest retrying can no longer consume
|
||||
// the allowance of everyone else sharing the venue's NAT.
|
||||
if rate_limits_on && join_rate_on {
|
||||
let name_key = display_name.to_lowercase();
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("join:{ip}:{name_key}"),
|
||||
5,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let event = Event::find_or_create(
|
||||
@@ -67,10 +144,21 @@ pub async fn join(
|
||||
|
||||
// Generate a 4-digit PIN
|
||||
let pin: String = format!("{:04}", rand::rng().random_range(0..10000u32));
|
||||
let pin_hash =
|
||||
bcrypt::hash(&pin, 12).map_err(|e| AppError::Internal(anyhow::anyhow!(e)))?;
|
||||
let pin_hash = hash_password(pin.clone(), 12).await?;
|
||||
|
||||
let user = User::create(&state.pool, event.id, display_name, &pin_hash).await?;
|
||||
// The pre-check above is racy: two simultaneous joins with the same name can both
|
||||
// pass it, and the DB's unique index then rejects the loser. Map that unique
|
||||
// violation to the same clean 409 the pre-check returns, not a generic 500.
|
||||
let user = match User::create(&state.pool, event.id, display_name, &pin_hash).await {
|
||||
Ok(u) => u,
|
||||
Err(sqlx::Error::Database(db)) if db.is_unique_violation() => {
|
||||
return Err(AppError::Conflict(format!(
|
||||
"Der Name \"{}\" ist bereits vergeben.",
|
||||
display_name
|
||||
)));
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
|
||||
let token = jwt::create_token(
|
||||
user.id,
|
||||
@@ -96,6 +184,28 @@ pub async fn join(
|
||||
))
|
||||
}
|
||||
|
||||
/// Default for `recover_name_rate_per_15min` — wrong PINs allowed per (IP, name) per 15 min.
|
||||
/// Mirrors migration 023; kept here so the invariant below can be asserted in a test.
|
||||
const RECOVER_NAME_CEILING_DEFAULT: usize = 4;
|
||||
|
||||
/// Wrong PINs, from ALL sources, before the ACCOUNT itself is locked for 15 minutes.
|
||||
///
|
||||
/// This was 3, which sat BELOW the per-(IP, name) ceiling of 5 — and that ordering, not the
|
||||
/// number, was the defect. Display names are public on the feed, so three requests from a single
|
||||
/// IP locked any guest out of their own account, repeatable every 15 minutes, indefinitely. The
|
||||
/// tier meant to protect a guest was the easiest way to attack them.
|
||||
///
|
||||
/// Raised deliberately far above the per-IP tier so the two do different jobs. The per-(IP, name)
|
||||
/// bucket is what stops a guesser, and it costs the ATTACKER. This tier is the last line against
|
||||
/// a DISTRIBUTED guesser, and it is the only one an attacker can turn on a victim — so reaching
|
||||
/// it must require at least three distinct sources inside the decay window.
|
||||
///
|
||||
/// Brute-force cost is unchanged: 12 attempts per 15 minutes is 48/hour against one account, so
|
||||
/// 10 000 four-digit PINs still take ~208 hours no matter how many IPs are used. An honest guest
|
||||
/// fat-fingering a 4-digit PIN never comes close, and `increment_failed_pin` now decays the
|
||||
/// streak after 15 minutes so yesterday's typos don't count toward today's.
|
||||
const PIN_LOCK_THRESHOLD: i16 = 12;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct RecoverRequest {
|
||||
pub display_name: String,
|
||||
@@ -108,38 +218,135 @@ pub struct RecoverResponse {
|
||||
pub user_id: Uuid,
|
||||
}
|
||||
|
||||
/// A real cost-12 bcrypt hash of a fixed dummy value, computed once and cached. Used
|
||||
/// to run a constant-time-ish verify on the "unknown display name" branch of
|
||||
/// [`recover`], so that branch costs the same as a real (user-exists) verify and can't
|
||||
/// be told apart by timing.
|
||||
fn dummy_pin_hash() -> &'static str {
|
||||
static HASH: std::sync::OnceLock<String> = std::sync::OnceLock::new();
|
||||
HASH.get_or_init(|| {
|
||||
bcrypt::hash("eventsnap-dummy-not-a-real-pin", 12)
|
||||
.expect("bcrypt hashing of a static dummy value cannot fail")
|
||||
})
|
||||
}
|
||||
|
||||
/// Run a bcrypt verify on the blocking pool.
|
||||
///
|
||||
/// bcrypt at cost 12 is ~200ms of deliberate CPU. Called inline on an async task it pins a
|
||||
/// tokio WORKER thread for that whole time, and the runtime only has one per core — so a
|
||||
/// flood of `/recover` or `/admin/login` attempts stalls every other request on the box,
|
||||
/// including the feed. Offloading moves that cost to the blocking pool, which is sized for
|
||||
/// exactly this and whose saturation degrades logins rather than the whole app.
|
||||
async fn verify_password(candidate: String, hash: String) -> bool {
|
||||
tokio::task::spawn_blocking(move || bcrypt::verify(&candidate, &hash).unwrap_or(false))
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Hash a secret on the blocking pool. Same reasoning as [`verify_password`] — and this one
|
||||
/// runs on the busiest auth path there is, since every guest who joins gets a PIN hashed.
|
||||
pub async fn hash_password(secret: String, cost: u32) -> Result<String, AppError> {
|
||||
tokio::task::spawn_blocking(move || bcrypt::hash(&secret, cost))
|
||||
.await
|
||||
.map_err(|e| AppError::Internal(anyhow::anyhow!(e)))?
|
||||
.map_err(|e| AppError::Internal(anyhow::anyhow!(e)))
|
||||
}
|
||||
|
||||
pub async fn recover(
|
||||
State(state): State<AppState>,
|
||||
ConnectInfo(peer): ConnectInfo<SocketAddr>,
|
||||
headers: HeaderMap,
|
||||
Json(body): Json<RecoverRequest>,
|
||||
) -> Result<Json<RecoverResponse>, AppError> {
|
||||
let display_name = body.display_name.trim();
|
||||
// Validated BEFORE it is used as a rate-limiter key — see `validate_display_name`. The
|
||||
// per-IP ceiling below is keyed only on the IP, so it is safe to run either side of this;
|
||||
// the per-NAME bucket is not.
|
||||
let display_name = validate_display_name(&body.display_name)?;
|
||||
|
||||
// Per-IP+name throttle BEFORE the per-user lockout counter. Without this an attacker who
|
||||
// knows a display name (they're visible on the feed) can burn through the victim's wrong-PIN
|
||||
// budget and lock them out, repeatedly. The ceiling here MUST stay below
|
||||
// PIN_LOCK_THRESHOLD — see the constant for why that ordering is the whole control.
|
||||
let ip = client_ip(&headers, &peer.ip().to_string());
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let recover_rate_on = config::get_bool(&state.config_cache, "recover_rate_enabled", true).await;
|
||||
if rate_limits_on && recover_rate_on {
|
||||
// Coarse per-IP ceiling FIRST. The per-(ip, name) bucket below is the anti-guessing
|
||||
// control, but the name is attacker-chosen, so cycling names mints a fresh bucket
|
||||
// every time and leaves the per-IP cost unbounded. That matters more here than
|
||||
// anywhere else: every call runs a cost-12 bcrypt verify, including an
|
||||
// unconditional throwaway one for names that don't exist (see below), so an unknown
|
||||
// name is the CHEAPEST way to make the server do ~200ms of hashing. Checked before
|
||||
// the per-name bucket so a name generator can't walk past it.
|
||||
let ip_ceiling =
|
||||
config::get_usize(&state.config_cache, "recover_ip_rate_per_min", 30).await;
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("recover_ip:{ip}"),
|
||||
ip_ceiling,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Versuche. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
|
||||
let name_ceiling = config::get_usize(
|
||||
&state.config_cache,
|
||||
"recover_name_rate_per_15min",
|
||||
RECOVER_NAME_CEILING_DEFAULT,
|
||||
)
|
||||
.await;
|
||||
let name_key = display_name.to_lowercase();
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("recover:{ip}:{name_key}"),
|
||||
name_ceiling,
|
||||
Duration::from_secs(15 * 60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Versuche. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let event = Event::find_by_slug(&state.pool, &state.config.event_slug)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
|
||||
let users =
|
||||
User::find_by_event_and_name(&state.pool, event.id, display_name).await?;
|
||||
let users = User::find_by_event_and_name(&state.pool, event.id, display_name).await?;
|
||||
|
||||
if users.is_empty() {
|
||||
return Err(AppError::NotFound(
|
||||
"Kein Benutzer mit diesem Namen gefunden.".into(),
|
||||
));
|
||||
// No user with this name. Run a throwaway bcrypt verify so this branch takes
|
||||
// the same time as the user-exists path, and return the SAME error as a wrong
|
||||
// PIN — so "no such name" and "wrong PIN" are indistinguishable by response or
|
||||
// timing. Display names are already public on the feed, but this still closes
|
||||
// the /recover enumeration + timing oracle.
|
||||
let _ = verify_password(body.pin.clone(), dummy_pin_hash().to_string()).await;
|
||||
return Err(AppError::Unauthorized("PIN ist falsch.".into()));
|
||||
}
|
||||
|
||||
for user in &users {
|
||||
// Check PIN lockout
|
||||
// Check PIN lockout. If the lockout has expired, also reset the failed-attempt
|
||||
// counter so the user gets a fresh 3-strike window — otherwise the counter
|
||||
// stays at 3+ and every subsequent wrong PIN immediately re-locks them, even
|
||||
// after waiting out the cooldown. Without this reset, a once-locked account
|
||||
// is effectively permanently fragile.
|
||||
if let Some(locked_until) = user.pin_locked_until {
|
||||
if Utc::now() < locked_until {
|
||||
// The exact deadline is known, so surface it as Retry-After instead of
|
||||
// making the client guess at the "15 Minuten" in the copy.
|
||||
let retry_after_secs = (locked_until - Utc::now()).num_seconds().max(1) as u64;
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Versuche. Bitte warte 15 Minuten.".into(),
|
||||
None,
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
// Lockout window expired — wipe the counter and the timestamp.
|
||||
User::reset_pin_attempts(&state.pool, user.id).await?;
|
||||
}
|
||||
|
||||
let pin_matches = bcrypt::verify(&body.pin, &user.recovery_pin_hash)
|
||||
.unwrap_or(false);
|
||||
let pin_matches = verify_password(body.pin.clone(), user.recovery_pin_hash.clone()).await;
|
||||
|
||||
if pin_matches {
|
||||
// Reset failed attempts on success
|
||||
@@ -166,9 +373,22 @@ pub async fn recover(
|
||||
|
||||
// Wrong PIN — increment failure count
|
||||
let attempts = User::increment_failed_pin(&state.pool, user.id).await?;
|
||||
if attempts >= 3 {
|
||||
tracing::warn!(
|
||||
user_id = %user.id,
|
||||
event_id = %event.id,
|
||||
ip = %ip,
|
||||
attempts,
|
||||
"recover: wrong PIN"
|
||||
);
|
||||
if attempts >= PIN_LOCK_THRESHOLD {
|
||||
let lockout = Utc::now() + chrono::Duration::minutes(15);
|
||||
User::lock_pin(&state.pool, user.id, lockout).await?;
|
||||
tracing::warn!(
|
||||
user_id = %user.id,
|
||||
event_id = %event.id,
|
||||
ip = %ip,
|
||||
"recover: account locked for 15 minutes"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -183,10 +403,16 @@ pub struct AdminLoginRequest {
|
||||
#[derive(Serialize)]
|
||||
pub struct AdminLoginResponse {
|
||||
pub jwt: String,
|
||||
/// The admin's user id + display name, so the client can populate a real identity
|
||||
/// (own-post affordances, a name on the Account page) instead of a blank session.
|
||||
pub user_id: Uuid,
|
||||
pub display_name: String,
|
||||
}
|
||||
|
||||
pub async fn admin_login(
|
||||
State(state): State<AppState>,
|
||||
ConnectInfo(peer): ConnectInfo<SocketAddr>,
|
||||
headers: HeaderMap,
|
||||
Json(body): Json<AdminLoginRequest>,
|
||||
) -> Result<Json<AdminLoginResponse>, AppError> {
|
||||
if state.config.admin_password_hash.is_empty() {
|
||||
@@ -195,10 +421,38 @@ pub async fn admin_login(
|
||||
));
|
||||
}
|
||||
|
||||
let valid = bcrypt::verify(&body.password, &state.config.admin_password_hash)
|
||||
.unwrap_or(false);
|
||||
// Throttle password attempts. The admin password is bcrypt-hashed (slow to
|
||||
// verify) but with no IP-level limit a determined attacker can still mount
|
||||
// a long-running guess campaign. 5 attempts / minute / IP is plenty for
|
||||
// honest typos.
|
||||
let ip = client_ip(&headers, &peer.ip().to_string());
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let admin_rate_on =
|
||||
config::get_bool(&state.config_cache, "admin_login_rate_enabled", true).await;
|
||||
// Stays keyed by IP on purpose: this guards a single shared credential, so a per-user
|
||||
// or per-name key would just hand an attacker a fresh bucket per guess.
|
||||
if rate_limits_on
|
||||
&& admin_rate_on
|
||||
&& let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("admin_login:{ip}"),
|
||||
5,
|
||||
Duration::from_secs(60),
|
||||
)
|
||||
{
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anmeldeversuche. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
|
||||
let valid = verify_password(
|
||||
body.password.clone(),
|
||||
state.config.admin_password_hash.clone(),
|
||||
)
|
||||
.await;
|
||||
|
||||
if !valid {
|
||||
tracing::warn!(ip = %ip, "admin_login: wrong password");
|
||||
return Err(AppError::Unauthorized("Falsches Passwort.".into()));
|
||||
}
|
||||
|
||||
@@ -209,25 +463,20 @@ pub async fn admin_login(
|
||||
)
|
||||
.await?;
|
||||
|
||||
// Find or create the admin user for this event
|
||||
let admin_name = "Admin";
|
||||
let users = User::find_by_event_and_name(&state.pool, event.id, admin_name).await?;
|
||||
let admin_user = if let Some(u) = users.into_iter().find(|u| u.role == UserRole::Admin) {
|
||||
u
|
||||
} else {
|
||||
// Create admin user with a dummy PIN (admin authenticates via password)
|
||||
let dummy_hash = bcrypt::hash("0000", 4)
|
||||
.map_err(|e| AppError::Internal(anyhow::anyhow!(e)))?;
|
||||
let user = User::create(&state.pool, event.id, admin_name, &dummy_hash).await?;
|
||||
sqlx::query("UPDATE \"user\" SET role = 'admin' WHERE id = $1")
|
||||
.bind(user.id)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
User::find_by_id(&state.pool, user.id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::Internal(anyhow::anyhow!("admin user creation failed")))?
|
||||
// Find or create the admin user for this event — BY ROLE, never by name.
|
||||
//
|
||||
// The name lookup this replaces is what made admin login brickable. Migration 007 makes
|
||||
// display_name unique per event case-insensitively and `join` had no reserved-name guard,
|
||||
// so a guest joining as "admin" before the operator's first login made the lookup miss on
|
||||
// role, the fallback `create("Admin")` violate that index, and `?` return a permanent 500 —
|
||||
// taking out moderation, config and gallery release with no in-app recovery.
|
||||
let admin_user = match User::find_admin_for_event(&state.pool, event.id).await? {
|
||||
Some(u) => u,
|
||||
None => create_admin_user(&state, event.id).await?,
|
||||
};
|
||||
|
||||
tracing::info!(user_id = %admin_user.id, event_id = %event.id, ip = %ip, "admin_login: success");
|
||||
|
||||
let token = jwt::create_token(
|
||||
admin_user.id,
|
||||
event.id,
|
||||
@@ -241,13 +490,219 @@ pub async fn admin_login(
|
||||
let expires_at = Utc::now() + chrono::Duration::days(1);
|
||||
Session::create(&state.pool, admin_user.id, &token_hash, expires_at).await?;
|
||||
|
||||
Ok(Json(AdminLoginResponse { jwt: token }))
|
||||
Ok(Json(AdminLoginResponse {
|
||||
jwt: token,
|
||||
user_id: admin_user.id,
|
||||
display_name: admin_user.display_name,
|
||||
}))
|
||||
}
|
||||
|
||||
pub async fn logout(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
/// Create this event's admin row on first successful admin login.
|
||||
///
|
||||
/// Prefers the name "Admin". If a legacy database has a guest squatting on it — the state
|
||||
/// migration 023 renames away, but a row could also predate that or be created between
|
||||
/// migrations — falls back to a suffixed name rather than failing the login.
|
||||
///
|
||||
/// PROMOTING THE SQUATTING ROW WOULD BE A SERIOUS MISTAKE, and is the obvious-looking fix, so
|
||||
/// it is spelled out: that row carries a `recovery_pin_hash` the guest knows. Setting
|
||||
/// `role = 'admin'` on it would hand them the admin dashboard through `/recover`, permanently,
|
||||
/// via a path that needs no password. A separate row under an uglier name is worse UX and much
|
||||
/// better security — and since the lookup is now by role, the fallback name never has to be
|
||||
/// guessed again on a later login.
|
||||
async fn create_admin_user(state: &AppState, event_id: Uuid) -> Result<User, AppError> {
|
||||
// Admin authenticates via password, but the schema still requires a PIN hash. Generate a
|
||||
// random unguessable one so the recovery path stays unusable as an escalation route even if
|
||||
// the role flag were ever cleared.
|
||||
let dummy_pin: String = (0..32)
|
||||
.map(|_| rand::rng().random_range(b'a'..=b'z') as char)
|
||||
.collect();
|
||||
let dummy_hash = hash_password(dummy_pin, 4).await?;
|
||||
|
||||
match User::create_with_role(&state.pool, event_id, "Admin", &dummy_hash, UserRole::Admin).await
|
||||
{
|
||||
Ok(u) => Ok(u),
|
||||
Err(sqlx::Error::Database(db)) if db.is_unique_violation() => {
|
||||
let fallback = format!("Admin-{}", &Uuid::new_v4().to_string()[..8]);
|
||||
tracing::warn!(
|
||||
%event_id, %fallback,
|
||||
"the name \"Admin\" is held by a non-admin user; creating the admin under a \
|
||||
fallback name. Rename that guest to free it — do NOT promote their row, they \
|
||||
know its recovery PIN."
|
||||
);
|
||||
Ok(User::create_with_role(
|
||||
&state.pool,
|
||||
event_id,
|
||||
&fallback,
|
||||
&dummy_hash,
|
||||
UserRole::Admin,
|
||||
)
|
||||
.await?)
|
||||
}
|
||||
Err(e) => Err(e.into()),
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn logout(State(state): State<AppState>, auth: AuthUser) -> Result<StatusCode, AppError> {
|
||||
Session::delete_by_token_hash(&state.pool, &auth.token_hash).await?;
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
/// "Sign out everywhere" — revoke every session for the caller, not just the current one.
|
||||
/// A single leaked/kept-alive device (a shared phone, a laptop recovered via PIN) can then
|
||||
/// be cut from any of the user's devices.
|
||||
pub async fn logout_all(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
Session::delete_all_for_user(&state.pool, auth.user_id).await?;
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct PinResetRequestBody {
|
||||
pub display_name: String,
|
||||
}
|
||||
|
||||
/// A guest who forgot their PIN asks a host to reset it in-app. Unauthenticated (they
|
||||
/// can't log in without the PIN) and rate-limited. Always returns 204 regardless of
|
||||
/// whether the name exists, so it can't enumerate display names beyond what the public
|
||||
/// feed already exposes.
|
||||
pub async fn request_pin_reset(
|
||||
State(state): State<AppState>,
|
||||
ConnectInfo(peer): ConnectInfo<SocketAddr>,
|
||||
headers: HeaderMap,
|
||||
Json(body): Json<PinResetRequestBody>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
let ip = client_ip(&headers, &peer.ip().to_string());
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
|
||||
// Coarse per-IP ceiling FIRST, keyed only on the IP so its key is bounded by construction.
|
||||
// /join got one of these in migration 017 and /recover in 019; this third unauthenticated
|
||||
// endpoint was simply missed — migration 019's own comment describes exactly this attack.
|
||||
// Without it, the per-name bucket below is no ceiling at all: the name is attacker-chosen,
|
||||
// so cycling names mints a fresh bucket every request.
|
||||
if rate_limits_on {
|
||||
let ip_ceiling =
|
||||
config::get_usize(&state.config_cache, "pin_reset_ip_rate_per_min", 30).await;
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("pin_reset_ip:{ip}"),
|
||||
ip_ceiling,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Validated BEFORE the per-name key is built, so an unbounded name can never be retained in
|
||||
// the limiter map. NOTE the 204: this endpoint's contract is that it answers identically
|
||||
// whether or not the name exists, so it cannot enumerate guests. A 400 here would be a new
|
||||
// signal — it would distinguish a malformed name from a well-formed unknown one. Silence is
|
||||
// the correct response, and matches what an empty name already did.
|
||||
let Ok(display_name) = validate_display_name(&body.display_name) else {
|
||||
return Ok(StatusCode::NO_CONTENT);
|
||||
};
|
||||
|
||||
if rate_limits_on {
|
||||
let name_key = display_name.to_lowercase();
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("pin_reset_req:{ip}:{name_key}"),
|
||||
3,
|
||||
Duration::from_secs(15 * 60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Single statement so the existing-name and unknown-name paths do IDENTICAL work
|
||||
// (same event+user index scans, an INSERT that matches 0 rows for an unknown name) —
|
||||
// no timing oracle despite the always-204 contract. Admins recover via password, so
|
||||
// they're excluded from the join and never get a reset request queued.
|
||||
let _ = sqlx::query(
|
||||
"INSERT INTO pin_reset_request (event_id, user_id)
|
||||
SELECT e.id, u.id
|
||||
FROM event e
|
||||
JOIN \"user\" u
|
||||
ON u.event_id = e.id
|
||||
AND lower(u.display_name) = lower($2)
|
||||
AND u.role <> 'admin'
|
||||
WHERE e.slug = $1
|
||||
ON CONFLICT (user_id) DO NOTHING",
|
||||
)
|
||||
.bind(&state.config.event_slug)
|
||||
.bind(display_name)
|
||||
.execute(&state.pool)
|
||||
.await;
|
||||
|
||||
// Broadcast unconditionally (carries no per-name info) so an online host's request
|
||||
// badge updates without revealing whether the name existed.
|
||||
let _ = state
|
||||
.sse_tx
|
||||
.send(crate::state::SseEvent::new("pin-reset-requested", "{}"));
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// THE defect, stated as arithmetic: the account-lock threshold sat BELOW the per-(IP, name)
|
||||
/// attempt ceiling, so a single IP could exhaust it and lock any guest whose display name is
|
||||
/// visible on the feed — every 15 minutes, indefinitely. The tier meant to protect a guest
|
||||
/// was the cheapest way to attack them.
|
||||
///
|
||||
/// The fix is the ORDERING, not either number on its own, so that is what this pins.
|
||||
#[test]
|
||||
fn one_ip_cannot_reach_the_account_lock() {
|
||||
assert!(
|
||||
PIN_LOCK_THRESHOLD as usize >= RECOVER_NAME_CEILING_DEFAULT * 3,
|
||||
"locking a victim must require at least three distinct sources; \
|
||||
threshold {PIN_LOCK_THRESHOLD} vs per-IP ceiling {RECOVER_NAME_CEILING_DEFAULT}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Raising the threshold must not quietly weaken brute-force resistance. 4-digit PINs, and
|
||||
/// the lockout window is 15 minutes, so an attacker gets PIN_LOCK_THRESHOLD tries per window.
|
||||
#[test]
|
||||
fn the_raised_threshold_still_makes_guessing_a_four_digit_pin_impractical() {
|
||||
let attempts_per_hour = PIN_LOCK_THRESHOLD as u64 * 4; // four 15-minute windows
|
||||
let hours_for_full_keyspace = 10_000 / attempts_per_hour;
|
||||
assert!(
|
||||
hours_for_full_keyspace >= 168,
|
||||
"exhausting 10k PINs would take {hours_for_full_keyspace}h — under a week is too fast"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reserved_names_are_matched_case_insensitively_and_trimmed() {
|
||||
for name in ["admin", "Admin", "ADMIN", " Host ", "EventSnap"] {
|
||||
assert!(is_reserved_display_name(name), "{name} must be reserved");
|
||||
}
|
||||
}
|
||||
|
||||
/// A substring match here would reject perfectly ordinary names, which is a worse outcome
|
||||
/// than the impersonation the list guards against.
|
||||
#[test]
|
||||
fn names_that_merely_contain_a_reserved_word_are_allowed() {
|
||||
for name in ["Administrata", "Hostess", "Adminah", "Ghost", "hosting"] {
|
||||
assert!(!is_reserved_display_name(name), "{name} must be allowed");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn display_names_are_bounded_before_they_can_become_a_rate_limit_key() {
|
||||
assert!(validate_display_name(" Lena ").is_ok());
|
||||
assert_eq!(validate_display_name(" Lena ").unwrap(), "Lena");
|
||||
// The case that made the limiter itself the exhaustion primitive.
|
||||
assert!(validate_display_name(&"a".repeat(51)).is_err());
|
||||
assert!(validate_display_name(&"a".repeat(2_000_000)).is_err());
|
||||
assert!(validate_display_name(" ").is_err());
|
||||
assert!(validate_display_name("bad\0name").is_err());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,13 @@ pub struct Claims {
|
||||
pub role: UserRole,
|
||||
pub exp: i64,
|
||||
pub iat: i64,
|
||||
/// Random per-token identifier. Without it, two `create_token` calls in the
|
||||
/// same wall-clock second for the same (sub, role, event) produce identical
|
||||
/// JWT bytes — and identical sha256(token) hashes — which then collide on
|
||||
/// the `session.token_hash` UNIQUE constraint. The jti is ignored by the
|
||||
/// verifier but breaks the collision.
|
||||
#[serde(default)]
|
||||
pub jti: Uuid,
|
||||
}
|
||||
|
||||
pub fn create_token(
|
||||
@@ -29,6 +36,7 @@ pub fn create_token(
|
||||
role,
|
||||
iat: now.timestamp(),
|
||||
exp: (now + Duration::days(expiry_days)).timestamp(),
|
||||
jti: Uuid::new_v4(),
|
||||
};
|
||||
jsonwebtoken::encode(
|
||||
&Header::default(),
|
||||
@@ -38,10 +46,20 @@ pub fn create_token(
|
||||
}
|
||||
|
||||
pub fn verify_token(token: &str, secret: &str) -> Result<Claims, jsonwebtoken::errors::Error> {
|
||||
// We deliberately do NOT enforce the JWT's own `exp`. The authoritative session
|
||||
// lifetime lives in the `session` row (`expires_at > NOW()`), which SLIDES forward on
|
||||
// every authenticated request (see `Session::touch_and_renew`). Enforcing the JWT's
|
||||
// fixed +Nd `exp` here would hard-log-out an actively-used client on day N+1 even
|
||||
// though its session was renewed — the "30-day cliff" from the review. With server-
|
||||
// side sliding sessions, the token is a signature-checked bearer credential and its
|
||||
// lifetime is revocable (logout / expiry / ban all delete the row), which is strictly
|
||||
// stronger than a stateless non-revocable exp.
|
||||
let mut validation = Validation::default();
|
||||
validation.validate_exp = false;
|
||||
let data = jsonwebtoken::decode::<Claims>(
|
||||
token,
|
||||
&DecodingKey::from_secret(secret.as_bytes()),
|
||||
&Validation::default(),
|
||||
&validation,
|
||||
)?;
|
||||
Ok(data.claims)
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use axum::extract::{FromRequestParts, State};
|
||||
use axum::extract::FromRequestParts;
|
||||
use axum::http::request::Parts;
|
||||
use uuid::Uuid;
|
||||
|
||||
@@ -13,6 +13,10 @@ pub struct AuthUser {
|
||||
pub user_id: Uuid,
|
||||
pub event_id: Uuid,
|
||||
pub role: UserRole,
|
||||
/// Live ban flag. Banned users keep *read* access (per USER_JOURNEYS §10), so
|
||||
/// the base extractor does NOT reject them — write handlers and the
|
||||
/// Require{Host,Admin} extractors enforce the ban instead.
|
||||
pub is_banned: bool,
|
||||
pub token_hash: String,
|
||||
}
|
||||
|
||||
@@ -33,27 +37,51 @@ impl FromRequestParts<AppState> for AuthUser {
|
||||
.strip_prefix("Bearer ")
|
||||
.ok_or_else(|| AppError::Unauthorized("Ungültiges Token-Format.".into()))?;
|
||||
|
||||
let claims = jwt::verify_token(token, &state.config.jwt_secret)
|
||||
// Verify the JWT's signature. Expiry is deliberately NOT enforced here (see
|
||||
// `jwt::verify_token`) — the authoritative, sliding session lifetime lives in the
|
||||
// `session` row read below. We also don't trust the token's role/ban claims; the
|
||||
// live user row is authoritative, so the decoded claims aren't needed beyond this.
|
||||
jwt::verify_token(token, &state.config.jwt_secret)
|
||||
.map_err(|_| AppError::Unauthorized("Token ungültig oder abgelaufen.".into()))?;
|
||||
|
||||
let token_hash = jwt::hash_token(token);
|
||||
|
||||
let session = Session::find_by_token_hash(&state.pool, &token_hash)
|
||||
// Single round-trip: resolve the session token to its *live* user row. A
|
||||
// role/ban stored in the token would survive a demote/ban for the full session
|
||||
// lifetime (up to 30d), so we always re-read the user (a demoted host loses host
|
||||
// powers immediately). We do NOT reject banned users here — they retain read
|
||||
// access by design; writes and host/admin actions enforce the ban downstream.
|
||||
let user = Session::find_user_by_token_hash(&state.pool, &token_hash)
|
||||
.await
|
||||
.map_err(|e| AppError::Internal(e.into()))?
|
||||
.ok_or_else(|| AppError::Unauthorized("Sitzung nicht gefunden oder abgelaufen.".into()))?;
|
||||
.ok_or_else(|| {
|
||||
AppError::Unauthorized("Sitzung nicht gefunden oder abgelaufen.".into())
|
||||
})?;
|
||||
|
||||
// Update last_seen_at in the background (fire-and-forget)
|
||||
// Touch last_seen_at AND slide the session's expiry forward (fire-and-forget), so
|
||||
// an active client's session renews instead of hitting the fixed 30-day cliff.
|
||||
// Admin sessions keep their tighter 1-day window (they renew on activity but still
|
||||
// lapse a day after the admin goes idle). Failures are non-fatal but worth
|
||||
// surfacing — silent swallowing hides DB connection pressure that would otherwise
|
||||
// be the first symptom of a real problem.
|
||||
let pool = state.pool.clone();
|
||||
let session_id = session.id;
|
||||
let touch_hash = token_hash.clone();
|
||||
let expiry_days = if user.role == UserRole::Admin {
|
||||
1
|
||||
} else {
|
||||
state.config.session_expiry_days
|
||||
};
|
||||
tokio::spawn(async move {
|
||||
let _ = Session::touch(&pool, session_id).await;
|
||||
if let Err(e) = Session::touch_and_renew(&pool, &touch_hash, expiry_days).await {
|
||||
tracing::warn!(error = ?e, "session touch/renew failed");
|
||||
}
|
||||
});
|
||||
|
||||
Ok(Self {
|
||||
user_id: claims.sub,
|
||||
event_id: claims.event_id,
|
||||
role: claims.role,
|
||||
user_id: user.id,
|
||||
event_id: user.event_id,
|
||||
role: user.role,
|
||||
is_banned: user.is_banned,
|
||||
token_hash,
|
||||
})
|
||||
}
|
||||
@@ -70,6 +98,9 @@ impl FromRequestParts<AppState> for RequireHost {
|
||||
state: &AppState,
|
||||
) -> Result<Self, Self::Rejection> {
|
||||
let auth = AuthUser::from_request_parts(parts, state).await?;
|
||||
if auth.is_banned {
|
||||
return Err(AppError::Forbidden("Du bist gesperrt.".into()));
|
||||
}
|
||||
match auth.role {
|
||||
UserRole::Host | UserRole::Admin => Ok(Self(auth)),
|
||||
_ => Err(AppError::Forbidden("Nur für Hosts und Admins.".into())),
|
||||
@@ -88,6 +119,9 @@ impl FromRequestParts<AppState> for RequireAdmin {
|
||||
state: &AppState,
|
||||
) -> Result<Self, Self::Rejection> {
|
||||
let auth = AuthUser::from_request_parts(parts, state).await?;
|
||||
if auth.is_banned {
|
||||
return Err(AppError::Forbidden("Du bist gesperrt.".into()));
|
||||
}
|
||||
match auth.role {
|
||||
UserRole::Admin => Ok(Self(auth)),
|
||||
_ => Err(AppError::Forbidden("Nur für Admins.".into())),
|
||||
|
||||
@@ -1,6 +1,83 @@
|
||||
use std::path::PathBuf;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
|
||||
/// Well-known dev JWT secret shipped in `.env.example`. If APP_ENV=production
|
||||
/// we refuse to start with this value; otherwise we warn loudly.
|
||||
const DEV_JWT_SECRET_SENTINEL: &str = "dev_secret_do_not_use_in_production_32byteslong_aaaa";
|
||||
|
||||
/// A secret is "placeholder-ish" if it's the shipped dev sentinel or still carries
|
||||
/// the tell-tale scaffolding substrings from `.env.example`. Length alone is not
|
||||
/// enough — the shipped `change_me_...` placeholder is >32 chars.
|
||||
fn looks_placeholder(s: &str) -> bool {
|
||||
let lower = s.to_ascii_lowercase();
|
||||
s == DEV_JWT_SECRET_SENTINEL
|
||||
|| lower.contains("change_me")
|
||||
|| lower.contains("dev_secret")
|
||||
|| lower.contains("placeholder")
|
||||
}
|
||||
|
||||
/// Enforce secret hygiene. In production every guard is hard-fail: a booting app
|
||||
/// with a publicly-known signing key is worse than one that refuses to start.
|
||||
/// Outside production the dev sentinel is tolerated (warned) so local dev is frictionless.
|
||||
///
|
||||
/// EVERY failure is collected and reported together. Returning on the first one made fixing two
|
||||
/// secrets cost two boot cycles — the operator rotates JWT_SECRET, restarts, and only then learns
|
||||
/// about ADMIN_PASSWORD_HASH. Restarting this stack is not free (Caddy waits on the unhealthy app),
|
||||
/// and each avoidable cycle is another chance to reach for `down -v`.
|
||||
fn validate_secrets(
|
||||
is_prod: bool,
|
||||
jwt_secret: &str,
|
||||
admin_password_hash: &str,
|
||||
database_url: &str,
|
||||
) -> Result<()> {
|
||||
if is_prod {
|
||||
let mut problems: Vec<&str> = Vec::new();
|
||||
if looks_placeholder(jwt_secret) {
|
||||
problems.push(
|
||||
"JWT_SECRET is still the .env.example placeholder — rotate it \
|
||||
(openssl rand -hex 64).",
|
||||
);
|
||||
} else if jwt_secret.len() < 32 {
|
||||
problems.push("JWT_SECRET must be at least 32 characters.");
|
||||
}
|
||||
if admin_password_hash.is_empty() || looks_placeholder(admin_password_hash) {
|
||||
problems.push(
|
||||
"ADMIN_PASSWORD_HASH is unset or still the .env.example placeholder — generate one \
|
||||
(docker run --rm caddy:2-alpine caddy hash-password --plaintext '<password>').",
|
||||
);
|
||||
}
|
||||
// The DATABASE_URL carries the Postgres password, so a placeholder here means the stack is
|
||||
// running on `CHANGE_ME_use_a_strong_password` — a credential published in the repo. The
|
||||
// app used to boot green on it, because this guard only ever covered the two secrets it
|
||||
// was written for and nothing else looked at POSTGRES_PASSWORD at all.
|
||||
//
|
||||
// Read POSTGRES_PASSWORD's docs before changing this: it is applied ONLY at initdb, so the
|
||||
// remedy is not "edit .env and restart" — see the 28P01 diagnostic in db.rs.
|
||||
if looks_placeholder(database_url) {
|
||||
problems.push(
|
||||
"DATABASE_URL still carries the .env.example placeholder password — set a strong \
|
||||
one (openssl rand -hex 24) in BOTH DATABASE_URL and POSTGRES_PASSWORD.",
|
||||
);
|
||||
}
|
||||
if !problems.is_empty() {
|
||||
return Err(anyhow!(
|
||||
"Refusing to start in production — {} secret(s) still unset or placeholder:\n - {}\n\
|
||||
ALL secrets must be set BEFORE the first `docker compose up -d`: Postgres bakes \
|
||||
POSTGRES_PASSWORD into its data directory on first boot and ignores later changes.",
|
||||
problems.len(),
|
||||
problems.join("\n - ")
|
||||
));
|
||||
}
|
||||
} else if jwt_secret == DEV_JWT_SECRET_SENTINEL {
|
||||
tracing::warn!(
|
||||
"JWT_SECRET is the dev sentinel — fine for local development, NEVER ship this."
|
||||
);
|
||||
} else if jwt_secret.len() < 32 {
|
||||
return Err(anyhow!("JWT_SECRET must be at least 32 characters."));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct AppConfig {
|
||||
@@ -11,33 +88,248 @@ pub struct AppConfig {
|
||||
pub event_name: String,
|
||||
pub event_slug: String,
|
||||
pub media_path: PathBuf,
|
||||
/// Where export archives are written. MUST be outside `media_path` — that
|
||||
/// directory is served by a public `ServeDir`, so a predictable archive name
|
||||
/// under it would leak the whole gallery to anonymous visitors.
|
||||
pub export_path: PathBuf,
|
||||
pub app_port: u16,
|
||||
/// Number of concurrent media compression workers (read once at boot).
|
||||
pub compression_concurrency: usize,
|
||||
/// Master switch for the comment feature (env `COMMENTS_ENABLED`, default true).
|
||||
/// When false the backend rejects new comments and the frontend hides the whole
|
||||
/// comment UI. Existing comments stay in the DB (hidden), so flipping it back
|
||||
/// restores them. Boot-time immutable, like `compression_concurrency`.
|
||||
pub comments_enabled: bool,
|
||||
/// Default colour theme, used as the fallback when the DB config keys are unset.
|
||||
/// Runtime overrides live in the `config` table (admin UI); these env vars only
|
||||
/// seed the initial default. `preset` is an id the frontend knows (e.g.
|
||||
/// "champagne-gold", "rose", … or "custom"); the two seeds are `#rrggbb` brand +
|
||||
/// accent colours the whole palette is derived from.
|
||||
pub default_theme_preset: String,
|
||||
pub default_theme_primary: String,
|
||||
pub default_theme_accent: String,
|
||||
}
|
||||
|
||||
/// The shipped default brand/accent seed (champagne gold — matches the hand-tuned
|
||||
/// ramp in tailwind-theme.css). Kept here so an unset env still yields the current look.
|
||||
const DEFAULT_THEME_SEED: &str = "#8a6a2b";
|
||||
|
||||
impl AppConfig {
|
||||
pub fn from_env() -> Result<Self> {
|
||||
let app_env = std::env::var("APP_ENV").unwrap_or_else(|_| "development".to_string());
|
||||
let is_prod = app_env.eq_ignore_ascii_case("production");
|
||||
|
||||
let jwt_secret = std::env::var("JWT_SECRET").context("JWT_SECRET must be set")?;
|
||||
let admin_password_hash = std::env::var("ADMIN_PASSWORD_HASH").unwrap_or_default();
|
||||
let database_url = std::env::var("DATABASE_URL").context("DATABASE_URL must be set")?;
|
||||
|
||||
validate_secrets(is_prod, &jwt_secret, &admin_password_hash, &database_url)?;
|
||||
|
||||
Ok(Self {
|
||||
database_url: std::env::var("DATABASE_URL")
|
||||
.context("DATABASE_URL must be set")?,
|
||||
jwt_secret: std::env::var("JWT_SECRET")
|
||||
.context("JWT_SECRET must be set")?,
|
||||
database_url,
|
||||
jwt_secret,
|
||||
session_expiry_days: std::env::var("SESSION_EXPIRY_DAYS")
|
||||
.unwrap_or_else(|_| "30".to_string())
|
||||
.parse()
|
||||
.context("SESSION_EXPIRY_DAYS must be a number")?,
|
||||
admin_password_hash: std::env::var("ADMIN_PASSWORD_HASH")
|
||||
.unwrap_or_default(),
|
||||
event_name: std::env::var("EVENT_NAME")
|
||||
.unwrap_or_else(|_| "EventSnap".to_string()),
|
||||
event_slug: std::env::var("EVENT_SLUG")
|
||||
.context("EVENT_SLUG must be set")?,
|
||||
admin_password_hash,
|
||||
event_name: std::env::var("EVENT_NAME").unwrap_or_else(|_| "EventSnap".to_string()),
|
||||
event_slug: std::env::var("EVENT_SLUG").context("EVENT_SLUG must be set")?,
|
||||
media_path: PathBuf::from(
|
||||
std::env::var("MEDIA_PATH").unwrap_or_else(|_| "/media".to_string()),
|
||||
),
|
||||
export_path: PathBuf::from(
|
||||
std::env::var("EXPORT_PATH").unwrap_or_else(|_| "/exports".to_string()),
|
||||
),
|
||||
app_port: std::env::var("APP_PORT")
|
||||
.unwrap_or_else(|_| "3000".to_string())
|
||||
.parse()
|
||||
.context("APP_PORT must be a number")?,
|
||||
compression_concurrency: std::env::var("COMPRESSION_WORKER_CONCURRENCY")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.filter(|&n| n >= 1)
|
||||
.unwrap_or(2),
|
||||
comments_enabled: std::env::var("COMMENTS_ENABLED")
|
||||
.map(|v| {
|
||||
!matches!(
|
||||
v.trim().to_ascii_lowercase().as_str(),
|
||||
"false" | "0" | "no" | "off"
|
||||
)
|
||||
})
|
||||
.unwrap_or(true),
|
||||
default_theme_preset: std::env::var("THEME_PRESET")
|
||||
.unwrap_or_else(|_| "champagne-gold".to_string()),
|
||||
default_theme_primary: std::env::var("THEME_PRIMARY")
|
||||
.unwrap_or_else(|_| DEFAULT_THEME_SEED.to_string()),
|
||||
default_theme_accent: std::env::var("THEME_ACCENT")
|
||||
.unwrap_or_else(|_| DEFAULT_THEME_SEED.to_string()),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const REAL_SECRET: &str = "a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6a7b8c9d0e1f2a3b4c5d6e7f8a9b0c1d2";
|
||||
const REAL_HASH: &str = "$2y$12$abcdefghijklmnopqrstuv.wxyzABCDEFGHIJKLMNOPQRSTUVWXYZ012";
|
||||
const REAL_DB_URL: &str = "postgres://eventsnap:7f3a9c1e5b2d8a4f@db:5432/eventsnap";
|
||||
|
||||
#[test]
|
||||
fn prod_rejects_shipped_placeholder_secret() {
|
||||
// The exact string shipped in `.env` — >32 chars, so it must be caught by
|
||||
// the substring guard, not the length check.
|
||||
let err = validate_secrets(
|
||||
true,
|
||||
"change_me_to_a_random_64_byte_hex_string",
|
||||
REAL_HASH,
|
||||
REAL_DB_URL,
|
||||
);
|
||||
assert!(
|
||||
err.is_err(),
|
||||
"placeholder JWT_SECRET must be rejected in prod"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prod_rejects_dev_sentinel_and_short_secret() {
|
||||
assert!(validate_secrets(true, DEV_JWT_SECRET_SENTINEL, REAL_HASH, REAL_DB_URL).is_err());
|
||||
assert!(validate_secrets(true, "tooshort", REAL_HASH, REAL_DB_URL).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prod_rejects_missing_or_placeholder_admin_hash() {
|
||||
assert!(validate_secrets(true, REAL_SECRET, "", REAL_DB_URL).is_err());
|
||||
assert!(
|
||||
validate_secrets(
|
||||
true,
|
||||
REAL_SECRET,
|
||||
"$2y$12$placeholder_replace_me",
|
||||
REAL_DB_URL
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
/// The stack used to come up GREEN on the database password published in the repo: this guard
|
||||
/// covered the two secrets it was written for, and nothing anywhere looked at the Postgres
|
||||
/// credential. README step 2 doesn't name POSTGRES_PASSWORD either, so following the
|
||||
/// documented procedure verbatim shipped it.
|
||||
#[test]
|
||||
fn prod_rejects_the_shipped_placeholder_database_password() {
|
||||
let shipped = "postgres://eventsnap:CHANGE_ME_use_a_strong_password@db:5432/eventsnap";
|
||||
let err = validate_secrets(true, REAL_SECRET, REAL_HASH, shipped).unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("DATABASE_URL"),
|
||||
"the refusal must name DATABASE_URL, not just fail: {err}"
|
||||
);
|
||||
// And it must point at the initdb trap, or the operator edits .env, restarts, and lands
|
||||
// in a permanent auth-failure loop instead.
|
||||
assert!(
|
||||
err.to_string().contains("POSTGRES_PASSWORD"),
|
||||
"the refusal must name POSTGRES_PASSWORD as the other half: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Every problem in ONE message. Reporting them one per boot made fixing two secrets cost two
|
||||
/// restart cycles, on a stack where Caddy waits on the unhealthy app the whole time.
|
||||
#[test]
|
||||
fn prod_reports_every_placeholder_at_once() {
|
||||
let err = validate_secrets(
|
||||
true,
|
||||
"change_me_to_a_random_64_byte_hex_string",
|
||||
"$2y$12$placeholder_replace_me",
|
||||
"postgres://eventsnap:CHANGE_ME_use_a_strong_password@db:5432/eventsnap",
|
||||
)
|
||||
.unwrap_err()
|
||||
.to_string();
|
||||
for expected in ["JWT_SECRET", "ADMIN_PASSWORD_HASH", "DATABASE_URL"] {
|
||||
assert!(err.contains(expected), "{expected} missing from: {err}");
|
||||
}
|
||||
assert!(
|
||||
err.contains("3 secret(s)"),
|
||||
"the count must match what is listed: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prod_accepts_real_secrets() {
|
||||
assert!(validate_secrets(true, REAL_SECRET, REAL_HASH, REAL_DB_URL).is_ok());
|
||||
}
|
||||
|
||||
/// A real password that happens to contain no placeholder substring must pass — including one
|
||||
/// with URL-ish punctuation, so the guard can't be mistaken for a URL validator.
|
||||
#[test]
|
||||
fn prod_accepts_a_real_database_url_with_awkward_punctuation() {
|
||||
assert!(
|
||||
validate_secrets(
|
||||
true,
|
||||
REAL_SECRET,
|
||||
REAL_HASH,
|
||||
"postgres://eventsnap:aB3%24xY9-_.qW@db:5432/eventsnap"
|
||||
)
|
||||
.is_ok()
|
||||
);
|
||||
}
|
||||
|
||||
/// The e2e stack runs without APP_ENV=production, so none of this applies there — but assert
|
||||
/// it, because a guard that tripped in e2e would be found the hard way.
|
||||
#[test]
|
||||
fn non_prod_ignores_a_placeholder_database_url() {
|
||||
assert!(
|
||||
validate_secrets(
|
||||
false,
|
||||
REAL_SECRET,
|
||||
"",
|
||||
"postgres://eventsnap:CHANGE_ME_use_a_strong_password@db:5432/eventsnap"
|
||||
)
|
||||
.is_ok()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_prod_tolerates_dev_sentinel() {
|
||||
assert!(validate_secrets(false, DEV_JWT_SECRET_SENTINEL, "", REAL_DB_URL).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_prod_still_rejects_short_non_sentinel_secret() {
|
||||
assert!(validate_secrets(false, "tooshort", "", REAL_DB_URL).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn placeholder_detection_is_case_insensitive() {
|
||||
// looks_placeholder lowercases before matching — an upper/mixed-case
|
||||
// placeholder must still be rejected in prod.
|
||||
assert!(
|
||||
validate_secrets(
|
||||
true,
|
||||
"CHANGE_ME_TO_A_RANDOM_64_BYTE_HEX_STRING",
|
||||
REAL_HASH,
|
||||
REAL_DB_URL
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
assert!(
|
||||
validate_secrets(
|
||||
true,
|
||||
REAL_SECRET,
|
||||
"$2Y$12$PLACEHOLDER_replace_me",
|
||||
REAL_DB_URL
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prod_len_boundary_at_32() {
|
||||
// Exactly 32 non-placeholder chars is the minimum accepted; 31 is rejected.
|
||||
const LEN_32: &str = "abcdefghijklmnopqrstuvwxyz012345";
|
||||
const LEN_31: &str = "abcdefghijklmnopqrstuvwxyz01234";
|
||||
assert_eq!(LEN_32.len(), 32);
|
||||
assert_eq!(LEN_31.len(), 31);
|
||||
assert!(validate_secrets(true, LEN_32, REAL_HASH, REAL_DB_URL).is_ok());
|
||||
assert!(validate_secrets(true, LEN_31, REAL_HASH, REAL_DB_URL).is_err());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,19 +1,84 @@
|
||||
use anyhow::{Context, Result};
|
||||
use sqlx::postgres::PgPoolOptions;
|
||||
use sqlx::PgPool;
|
||||
use sqlx::postgres::PgPoolOptions;
|
||||
|
||||
const DEFAULT_MAX_CONNECTIONS: u32 = 10;
|
||||
|
||||
/// How long a request waits for a pool connection before being shed.
|
||||
///
|
||||
/// sqlx's default is 30 s — longer than the frontend's own 20 s fetch timeout (api.ts
|
||||
/// TIMEOUT_MS), so under saturation the browser gave up while the server kept holding the
|
||||
/// slot: the client saw a timeout, the server saw a completed request, and the work was done
|
||||
/// for nobody. Failing fast sheds load instead of compounding it, and `From<sqlx::Error>` in
|
||||
/// error.rs turns the timeout into a 503 with a Retry-After rather than a bare 500.
|
||||
const ACQUIRE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5);
|
||||
|
||||
/// SQLSTATE for `invalid_password`.
|
||||
const PG_INVALID_PASSWORD: &str = "28P01";
|
||||
|
||||
/// Turn the one connect failure with an unguessable cause into a self-explaining one.
|
||||
///
|
||||
/// `POSTGRES_PASSWORD` is honoured ONLY when Postgres initialises its data directory. Change it in
|
||||
/// `.env` afterwards and the app authenticates with the new password against a volume that still
|
||||
/// holds the old one — a permanent restart loop whose only symptom is
|
||||
/// `password authentication failed`.
|
||||
///
|
||||
/// The production secret guard makes that sequence NEARLY CERTAIN rather than rare: it stops the
|
||||
/// app on the first `docker compose up -d`, but not the `db` service in that same command, which
|
||||
/// initialises and bakes in whatever password was in `.env` at that moment. So the intended
|
||||
/// recovery — see the refusal, fix your secrets, boot again — is exactly the sequence that breaks
|
||||
/// it. Nothing in the error names the cause, and the remedy destroys data, so it is the last thing
|
||||
/// an operator should guess at.
|
||||
fn explain_auth_failure(err: &sqlx::Error) {
|
||||
let is_auth_failure = match err {
|
||||
sqlx::Error::Database(db) => db.code().as_deref() == Some(PG_INVALID_PASSWORD),
|
||||
_ => false,
|
||||
};
|
||||
if !is_auth_failure {
|
||||
return;
|
||||
}
|
||||
tracing::error!(
|
||||
"Postgres rejected the credentials in DATABASE_URL (SQLSTATE {PG_INVALID_PASSWORD}).\n\
|
||||
\n\
|
||||
This almost always means POSTGRES_PASSWORD was changed AFTER the database volume was \
|
||||
first created. Postgres applies that variable only when it initialises its data \
|
||||
directory; editing .env and restarting does not change the stored password, so the two \
|
||||
drift apart permanently.\n\
|
||||
\n\
|
||||
If the event has NOT started and you have no data worth keeping:\n\n \
|
||||
docker compose down -v && docker compose up -d\n\n\
|
||||
(-v DELETES the database, the uploaded media and the exports. There is no undo.)\n\
|
||||
\n\
|
||||
If you DO have data: restore the old password into DATABASE_URL instead, or change the \
|
||||
stored one with ALTER ROLE inside the running db container. Never reach for -v to fix a \
|
||||
login problem on a live event."
|
||||
);
|
||||
}
|
||||
|
||||
pub async fn create_pool(database_url: &str) -> Result<PgPool> {
|
||||
let pool = PgPoolOptions::new()
|
||||
.max_connections(10)
|
||||
let max_connections = std::env::var("DATABASE_MAX_CONNECTIONS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse::<u32>().ok())
|
||||
.unwrap_or(DEFAULT_MAX_CONNECTIONS);
|
||||
|
||||
let pool = match PgPoolOptions::new()
|
||||
.max_connections(max_connections)
|
||||
.acquire_timeout(ACQUIRE_TIMEOUT)
|
||||
.connect(database_url)
|
||||
.await
|
||||
.context("failed to connect to database")?;
|
||||
{
|
||||
Ok(pool) => pool,
|
||||
Err(e) => {
|
||||
explain_auth_failure(&e);
|
||||
return Err(e).context("failed to connect to database");
|
||||
}
|
||||
};
|
||||
|
||||
sqlx::migrate!()
|
||||
.run(&pool)
|
||||
.await
|
||||
.context("failed to run database migrations")?;
|
||||
|
||||
tracing::info!("database connected and migrations applied");
|
||||
tracing::info!(max_connections, "database connected and migrations applied");
|
||||
Ok(pool)
|
||||
}
|
||||
|
||||
@@ -6,10 +6,25 @@ pub enum AppError {
|
||||
BadRequest(String),
|
||||
Unauthorized(String),
|
||||
Forbidden(String),
|
||||
/// Uploads are temporarily locked (event closed / gallery released). Distinct from
|
||||
/// `Forbidden` so the client can tell this REVERSIBLE 403 apart from a permanent one
|
||||
/// (banned user, quota): the queued blob is kept and retried if the host reopens,
|
||||
/// instead of being purged like a genuinely-terminal rejection.
|
||||
UploadsLocked(String),
|
||||
NotFound(String),
|
||||
Conflict(String),
|
||||
/// Second field: optional retry-after seconds to include in the response.
|
||||
TooManyRequests(String, Option<u64>),
|
||||
/// Per-user storage quota exhausted. Distinct from `TooManyRequests` (rate limit) so
|
||||
/// the client can treat it as *terminal* (413, no retry) instead of backing off and
|
||||
/// retrying a permanently-failing upload forever.
|
||||
QuotaExceeded(String),
|
||||
/// The server is temporarily unable to serve this request — currently only pool
|
||||
/// saturation. Distinct from `Internal` because it is TRANSIENT and the client should be
|
||||
/// told so: a 500 reads as "this request is broken", while a 503 + Retry-After reads as
|
||||
/// "come back shortly", which is what the upload queue's retry classifier needs to make
|
||||
/// the right call. Second field: optional retry-after seconds.
|
||||
ServiceUnavailable(String, Option<u64>),
|
||||
Internal(anyhow::Error),
|
||||
}
|
||||
|
||||
@@ -19,9 +34,14 @@ impl AppError {
|
||||
Self::BadRequest(_) => (StatusCode::BAD_REQUEST, "bad_request"),
|
||||
Self::Unauthorized(_) => (StatusCode::UNAUTHORIZED, "unauthorized"),
|
||||
Self::Forbidden(_) => (StatusCode::FORBIDDEN, "forbidden"),
|
||||
Self::UploadsLocked(_) => (StatusCode::FORBIDDEN, "uploads_locked"),
|
||||
Self::NotFound(_) => (StatusCode::NOT_FOUND, "not_found"),
|
||||
Self::Conflict(_) => (StatusCode::CONFLICT, "conflict"),
|
||||
Self::TooManyRequests(..) => (StatusCode::TOO_MANY_REQUESTS, "too_many_requests"),
|
||||
Self::QuotaExceeded(_) => (StatusCode::PAYLOAD_TOO_LARGE, "quota_exceeded"),
|
||||
Self::ServiceUnavailable(..) => {
|
||||
(StatusCode::SERVICE_UNAVAILABLE, "service_unavailable")
|
||||
}
|
||||
Self::Internal(_) => (StatusCode::INTERNAL_SERVER_ERROR, "internal_error"),
|
||||
}
|
||||
}
|
||||
@@ -31,9 +51,12 @@ impl AppError {
|
||||
Self::BadRequest(msg)
|
||||
| Self::Unauthorized(msg)
|
||||
| Self::Forbidden(msg)
|
||||
| Self::UploadsLocked(msg)
|
||||
| Self::NotFound(msg)
|
||||
| Self::Conflict(msg) => msg.clone(),
|
||||
Self::TooManyRequests(msg, _) => msg.clone(),
|
||||
Self::ServiceUnavailable(msg, _) => msg.clone(),
|
||||
Self::QuotaExceeded(msg) => msg.clone(),
|
||||
Self::Internal(err) => {
|
||||
tracing::error!("internal error: {err:#}");
|
||||
"Ein interner Fehler ist aufgetreten.".to_string()
|
||||
@@ -45,10 +68,13 @@ impl AppError {
|
||||
impl IntoResponse for AppError {
|
||||
fn into_response(self) -> Response {
|
||||
let (status, code) = self.status_and_code();
|
||||
let retry_after_secs = if let Self::TooManyRequests(_, Some(secs)) = &self {
|
||||
Some(*secs)
|
||||
} else {
|
||||
None
|
||||
// BOTH retry-carrying variants must be matched here. `message()` would fail to
|
||||
// compile on a missing arm; this one would not — it would silently drop the header and
|
||||
// the `retry_after_secs` body field, which is exactly the sort of omission that only
|
||||
// shows up under the load the 503 exists for.
|
||||
let retry_after_secs = match &self {
|
||||
Self::TooManyRequests(_, secs) | Self::ServiceUnavailable(_, secs) => *secs,
|
||||
_ => None,
|
||||
};
|
||||
let message = self.message();
|
||||
|
||||
@@ -62,10 +88,11 @@ impl IntoResponse for AppError {
|
||||
}
|
||||
|
||||
let mut resp = (status, axum::Json(body)).into_response();
|
||||
if let Some(secs) = retry_after_secs {
|
||||
if let Ok(val) = axum::http::HeaderValue::from_str(&secs.to_string()) {
|
||||
resp.headers_mut().insert(axum::http::header::RETRY_AFTER, val);
|
||||
}
|
||||
if let Some(secs) = retry_after_secs
|
||||
&& let Ok(val) = axum::http::HeaderValue::from_str(&secs.to_string())
|
||||
{
|
||||
resp.headers_mut()
|
||||
.insert(axum::http::header::RETRY_AFTER, val);
|
||||
}
|
||||
resp
|
||||
}
|
||||
@@ -79,6 +106,84 @@ impl From<anyhow::Error> for AppError {
|
||||
|
||||
impl From<sqlx::Error> for AppError {
|
||||
fn from(err: sqlx::Error) -> Self {
|
||||
Self::Internal(err.into())
|
||||
match err {
|
||||
// Pool saturation is load, not a bug. Reporting it as a 500 was actively harmful:
|
||||
// the frontend's upload-queue classifier treats 5xx as transient and retries, so
|
||||
// the retries piled straight back into the saturated pool with no Retry-After to
|
||||
// pace them. A 503 says the same thing honestly and carries the backoff.
|
||||
//
|
||||
// `PoolClosed` stays `Internal` — it only happens during shutdown, where a 503
|
||||
// would invite a retry against a server that is going away.
|
||||
sqlx::Error::PoolTimedOut => {
|
||||
tracing::warn!("database pool exhausted; shedding a request with 503");
|
||||
Self::ServiceUnavailable(
|
||||
"Server ist gerade ausgelastet. Bitte versuche es in ein paar Sekunden erneut."
|
||||
.into(),
|
||||
Some(POOL_TIMEOUT_RETRY_AFTER_SECS),
|
||||
)
|
||||
}
|
||||
other => Self::Internal(other.into()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retry-After for a shed request. Short: pool saturation clears in seconds once the queue
|
||||
/// drains, and a long value would make a brief spike feel like an outage.
|
||||
const POOL_TIMEOUT_RETRY_AFTER_SECS: u64 = 3;
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// `into_response` extracts `retry_after_secs` by MATCHING ON VARIANTS, so unlike
|
||||
/// `message()` a missing arm is not a compile error — it silently drops the header. Pin the
|
||||
/// behaviour for both retry-carrying variants.
|
||||
#[test]
|
||||
fn both_retry_carrying_variants_emit_retry_after() {
|
||||
for err in [
|
||||
AppError::TooManyRequests("slow down".into(), Some(42)),
|
||||
AppError::ServiceUnavailable("busy".into(), Some(3)),
|
||||
] {
|
||||
let expected = match &err {
|
||||
AppError::TooManyRequests(_, Some(s)) | AppError::ServiceUnavailable(_, Some(s)) => {
|
||||
s.to_string()
|
||||
}
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let resp = err.into_response();
|
||||
assert_eq!(
|
||||
resp.headers()
|
||||
.get(axum::http::header::RETRY_AFTER)
|
||||
.and_then(|v| v.to_str().ok()),
|
||||
Some(expected.as_str()),
|
||||
"a shed/throttled client must be told when to come back"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Pool saturation is load, not a bug. A 500 makes the frontend's retry classifier pile
|
||||
/// straight back into the saturated pool with no backoff to pace it.
|
||||
#[test]
|
||||
fn pool_exhaustion_sheds_with_503_but_shutdown_does_not() {
|
||||
let shed: AppError = sqlx::Error::PoolTimedOut.into();
|
||||
assert_eq!(
|
||||
shed.status_and_code(),
|
||||
(StatusCode::SERVICE_UNAVAILABLE, "service_unavailable")
|
||||
);
|
||||
|
||||
// PoolClosed only happens during shutdown; a 503 there would invite a retry against a
|
||||
// server that is going away.
|
||||
let closing: AppError = sqlx::Error::PoolClosed.into();
|
||||
assert_eq!(
|
||||
closing.status_and_code(),
|
||||
(StatusCode::INTERNAL_SERVER_ERROR, "internal_error")
|
||||
);
|
||||
|
||||
// Everything else must keep its existing mapping.
|
||||
let missing: AppError = sqlx::Error::RowNotFound.into();
|
||||
assert_eq!(
|
||||
missing.status_and_code(),
|
||||
(StatusCode::INTERNAL_SERVER_ERROR, "internal_error")
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,15 +1,15 @@
|
||||
use std::collections::HashMap;
|
||||
use std::time::Duration;
|
||||
|
||||
use axum::extract::State;
|
||||
use axum::http::{HeaderMap, StatusCode};
|
||||
use axum::Json;
|
||||
use axum::extract::{Query, State};
|
||||
use axum::http::StatusCode;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use sysinfo::System;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::auth::middleware::RequireAdmin;
|
||||
use crate::error::AppError;
|
||||
use crate::services::rate_limiter::client_ip;
|
||||
use crate::services::config;
|
||||
use crate::state::AppState;
|
||||
|
||||
// ── DTOs ─────────────────────────────────────────────────────────────────────
|
||||
@@ -45,19 +45,17 @@ pub async fn get_stats(
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
|
||||
let (user_count,): (i64,) =
|
||||
sqlx::query_as("SELECT COUNT(*) FROM \"user\" WHERE event_id = $1")
|
||||
let (user_count,): (i64,) = sqlx::query_as("SELECT COUNT(*) FROM \"user\" WHERE event_id = $1")
|
||||
.bind(event.id)
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
|
||||
let (upload_count,): (i64,) =
|
||||
sqlx::query_as("SELECT COUNT(*) FROM upload WHERE event_id = $1 AND deleted_at IS NULL")
|
||||
.bind(event.id)
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
|
||||
let (upload_count,): (i64,) = sqlx::query_as(
|
||||
"SELECT COUNT(*) FROM upload WHERE event_id = $1 AND deleted_at IS NULL",
|
||||
)
|
||||
.bind(event.id)
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
|
||||
let (comment_count,): (i64,) = sqlx::query_as(
|
||||
"SELECT COUNT(*) FROM comment c
|
||||
JOIN upload u ON u.id = c.upload_id
|
||||
@@ -67,23 +65,12 @@ pub async fn get_stats(
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
|
||||
// Disk usage via sysinfo
|
||||
let mut sys = System::new();
|
||||
sys.refresh_all();
|
||||
|
||||
let media_path = state.config.media_path.to_string_lossy().to_string();
|
||||
let (disk_total, disk_free) = sysinfo::Disks::new_with_refreshed_list()
|
||||
.iter()
|
||||
.find(|d| media_path.starts_with(d.mount_point().to_string_lossy().as_ref()))
|
||||
.map(|d| (d.total_space(), d.available_space()))
|
||||
.unwrap_or_else(|| {
|
||||
// Fall back to the root disk
|
||||
sysinfo::Disks::new_with_refreshed_list()
|
||||
.iter()
|
||||
.find(|d| d.mount_point().to_string_lossy() == "/")
|
||||
.map(|d| (d.total_space(), d.available_space()))
|
||||
.unwrap_or((0, 0))
|
||||
});
|
||||
// Disk usage from the shared cache (unknown mount → zeros, same as before).
|
||||
let (disk_total, disk_free) = state
|
||||
.disk_cache
|
||||
.snapshot(&state.config.media_path)
|
||||
.map(|d| (d.total, d.free))
|
||||
.unwrap_or((0, 0));
|
||||
|
||||
let disk_used = disk_total.saturating_sub(disk_free);
|
||||
|
||||
@@ -101,14 +88,17 @@ pub async fn get_config(
|
||||
State(state): State<AppState>,
|
||||
RequireAdmin(_auth): RequireAdmin,
|
||||
) -> Result<Json<HashMap<String, String>>, AppError> {
|
||||
let rows: Vec<(String, String)> =
|
||||
sqlx::query_as("SELECT key, value FROM config ORDER BY key")
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
let rows: Vec<(String, String)> = sqlx::query_as("SELECT key, value FROM config ORDER BY key")
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
|
||||
Ok(Json(rows.into_iter().collect()))
|
||||
}
|
||||
|
||||
/// Documents the wire shape of `PATCH /admin/config` (a flat `{key: value}` object).
|
||||
/// `patch_config` extracts the `HashMap` directly rather than going through this newtype, so it is
|
||||
/// never constructed in Rust — it stays as the serde-derived description of the request body.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Deserialize)]
|
||||
pub struct PatchConfigRequest(pub HashMap<String, String>);
|
||||
|
||||
@@ -117,38 +107,200 @@ pub async fn patch_config(
|
||||
RequireAdmin(_auth): RequireAdmin,
|
||||
Json(body): Json<HashMap<String, String>>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
const ALLOWED_KEYS: &[&str] = &[
|
||||
"max_image_size_mb",
|
||||
"max_video_size_mb",
|
||||
"upload_rate_per_hour",
|
||||
"feed_rate_per_min",
|
||||
"export_rate_per_day",
|
||||
"quota_tolerance",
|
||||
"estimated_guest_count",
|
||||
"compression_concurrency",
|
||||
// Numeric keys validated as f64; boolean keys validated as truthy strings; the
|
||||
// privacy note is free text. Splitting these explicitly is verbose but makes the
|
||||
// failure mode for typos obvious (`Unbekannter Schlüssel: ...`).
|
||||
// (key, integer_only, min, max). Ranges reject values that `parse::<f64>` would
|
||||
// accept but that silently revert to the hardcoded default at read time
|
||||
// (get_usize/get_i64 can't parse negatives/NaN/fractionals). `compression_concurrency`
|
||||
// is intentionally absent — it's read once at boot, so a live edit was a no-op.
|
||||
const NUMERIC_SPECS: &[(&str, bool, f64, f64)] = &[
|
||||
("max_image_size_mb", true, 1.0, 1024.0),
|
||||
("max_video_size_mb", true, 1.0, 10240.0),
|
||||
("upload_rate_per_hour", true, 1.0, 100_000.0),
|
||||
("feed_rate_per_min", true, 1.0, 100_000.0),
|
||||
("export_rate_per_day", true, 1.0, 100_000.0),
|
||||
// Loose per-IP ceiling on /join. The real anti-spam bucket is per (ip, name); this
|
||||
// only bounds raw volume from one source, so it must stay well above the size of a
|
||||
// party arriving at once (see migration 017).
|
||||
("join_ip_rate_per_min", true, 1.0, 100_000.0),
|
||||
// Same shape for /recover: the per-(ip, name) bucket is the anti-guessing control,
|
||||
// this only bounds a name-cycling flood in front of a cost-12 bcrypt (migration 019).
|
||||
("recover_ip_rate_per_min", true, 1.0, 100_000.0),
|
||||
// Aggregate ceiling on likes + comments + comment deletions, per user per minute.
|
||||
// These were the only mutating endpoints with no limit at all (migration 020).
|
||||
("social_rate_per_min", true, 1.0, 100_000.0),
|
||||
("quota_tolerance", false, 0.0, 1.0),
|
||||
("estimated_guest_count", true, 1.0, 1_000_000.0),
|
||||
];
|
||||
const BOOL_KEYS: &[&str] = &[
|
||||
"rate_limits_enabled",
|
||||
"upload_rate_enabled",
|
||||
"feed_rate_enabled",
|
||||
"export_rate_enabled",
|
||||
"join_rate_enabled",
|
||||
// These two per-area rate toggles are HONOURED by their handlers (auth/handlers.rs reads
|
||||
// `admin_login_rate_enabled` and `recover_rate_enabled`, both defaulting true) but were
|
||||
// missing from this allowlist — so the switch existed in code and could never be flipped.
|
||||
"admin_login_rate_enabled",
|
||||
"recover_rate_enabled",
|
||||
"social_rate_enabled",
|
||||
"quota_enabled",
|
||||
"storage_quota_enabled",
|
||||
"upload_count_quota_enabled",
|
||||
];
|
||||
const TEXT_KEYS: &[&str] = &[
|
||||
"privacy_note",
|
||||
"theme_preset",
|
||||
"theme_primary",
|
||||
"theme_accent",
|
||||
];
|
||||
const PRIVACY_NOTE_MAX_LEN: usize = 16 * 1024; // 16 KiB free text is plenty
|
||||
// Preset ids the frontend knows how to render (mirror of PRESETS in
|
||||
// frontend/src/lib/theme/palette.ts). "custom" means "use the theme_primary/accent
|
||||
// seeds verbatim". Kept in sync by hand — a new preset must be added in both places.
|
||||
const THEME_PRESETS: &[&str] = &[
|
||||
"champagne-gold",
|
||||
"rose",
|
||||
"sage",
|
||||
"dusk-blue",
|
||||
"classic-silver",
|
||||
"custom",
|
||||
];
|
||||
|
||||
let mut privacy_note_changed = false;
|
||||
let mut theme_changed = false;
|
||||
|
||||
// Validate every key first so a bad value in the batch can't leave a partial
|
||||
// update behind — validation must fully precede any write.
|
||||
for (key, value) in &body {
|
||||
if !ALLOWED_KEYS.contains(&key.as_str()) {
|
||||
return Err(AppError::BadRequest(format!("Unbekannter Konfigurationsschlüssel: {key}")));
|
||||
}
|
||||
// Validate numeric values
|
||||
if value.parse::<f64>().is_err() {
|
||||
return Err(AppError::BadRequest(format!("Ungültiger Wert für {key}: muss eine Zahl sein.")));
|
||||
let key_str = key.as_str();
|
||||
if let Some(&(_, integer_only, min, max)) =
|
||||
NUMERIC_SPECS.iter().find(|(k, ..)| *k == key_str)
|
||||
{
|
||||
let n = value.trim().parse::<f64>().ok().filter(|n| n.is_finite());
|
||||
let n = match n {
|
||||
Some(n) => n,
|
||||
None => {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Ungültiger Wert für {key}: muss eine Zahl sein."
|
||||
)));
|
||||
}
|
||||
};
|
||||
if integer_only && n.fract() != 0.0 {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Ungültiger Wert für {key}: muss eine ganze Zahl sein."
|
||||
)));
|
||||
}
|
||||
if n < min || n > max {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Wert für {key} liegt außerhalb des zulässigen Bereichs ({min}–{max})."
|
||||
)));
|
||||
}
|
||||
// Zero is in range and catastrophic. `quota_tolerance` is the multiplier in
|
||||
// `free_disk * tolerance / active_uploaders`, so 0 makes every per-user limit 0 and
|
||||
// refuses EVERY upload — mid-event, with "Du hast dein Upload-Limit für dieses Event
|
||||
// erreicht", an error naming the wrong cause entirely. `storage_quota_enabled` is the
|
||||
// intended off-switch.
|
||||
//
|
||||
// Rejecting the value rather than raising the floor: very small tolerances are
|
||||
// legitimate (they are how a large disk is throttled down to a sensible per-guest
|
||||
// ceiling, and how the e2e quota tests steer it — around 1e-5 on a 174 GB volume), so
|
||||
// a floor of, say, 0.01 would forbid real configurations to prevent one typo.
|
||||
if key_str == "quota_tolerance" && n == 0.0 {
|
||||
return Err(AppError::BadRequest(
|
||||
"quota_tolerance = 0 würde jeden Upload blockieren. Zum Abschalten der \
|
||||
Speicher-Quote stattdessen „Speicher-Quote aktiv“ ausschalten."
|
||||
.into(),
|
||||
));
|
||||
}
|
||||
} else if BOOL_KEYS.contains(&key_str) {
|
||||
match value.trim().to_ascii_lowercase().as_str() {
|
||||
"true" | "false" | "1" | "0" | "yes" | "no" | "on" | "off" => {}
|
||||
_ => {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Ungültiger Wert für {key}: muss true oder false sein."
|
||||
)));
|
||||
}
|
||||
}
|
||||
} else if TEXT_KEYS.contains(&key_str) {
|
||||
// Count characters, not bytes — the message says "Zeichen" and a
|
||||
// multi-byte grapheme shouldn't count against the limit multiple times.
|
||||
if value.chars().count() > PRIVACY_NOTE_MAX_LEN {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Wert für {key} ist zu lang (max. {PRIVACY_NOTE_MAX_LEN} Zeichen)."
|
||||
)));
|
||||
}
|
||||
match key_str {
|
||||
"privacy_note" => privacy_note_changed = true,
|
||||
"theme_preset" => {
|
||||
if !THEME_PRESETS.contains(&value.trim()) {
|
||||
return Err(AppError::BadRequest(format!("Ungültiges Theme: {value}.")));
|
||||
}
|
||||
theme_changed = true;
|
||||
}
|
||||
"theme_primary" | "theme_accent" => {
|
||||
if !is_hex_color(value.trim()) {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Ungültige Farbe für {key}: muss #rrggbb sein."
|
||||
)));
|
||||
}
|
||||
theme_changed = true;
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
} else {
|
||||
return Err(AppError::BadRequest(format!(
|
||||
"Unbekannter Konfigurationsschlüssel: {key}"
|
||||
)));
|
||||
}
|
||||
}
|
||||
|
||||
// Apply all writes in one transaction — the batch is all-or-nothing.
|
||||
let mut tx = state.pool.begin().await?;
|
||||
for (key, value) in &body {
|
||||
sqlx::query(
|
||||
"INSERT INTO config (key, value, updated_at) VALUES ($1, $2, NOW())
|
||||
ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value, updated_at = NOW()",
|
||||
)
|
||||
.bind(key)
|
||||
.bind(value)
|
||||
.execute(&state.pool)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
}
|
||||
tx.commit().await?;
|
||||
|
||||
// The config cache must reflect this write on the very next read (tests PATCH then
|
||||
// immediately assert the new value takes effect). Invalidate synchronously here —
|
||||
// the TTL is only a backstop and must not be relied on for correctness.
|
||||
state.config_cache.invalidate();
|
||||
|
||||
// Notify all clients that a publicly-readable config value changed so their stores
|
||||
// (e.g. the privacy note in My Account) refresh without a manual reload.
|
||||
if privacy_note_changed || theme_changed {
|
||||
let mut keys: Vec<&str> = Vec::new();
|
||||
if privacy_note_changed {
|
||||
keys.push("privacy_note");
|
||||
}
|
||||
if theme_changed {
|
||||
keys.push("theme");
|
||||
}
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent::new(
|
||||
"event-updated",
|
||||
serde_json::json!({ "keys": keys }).to_string(),
|
||||
));
|
||||
}
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
/// A strict `#rrggbb` hex-colour check (6 hex digits, leading `#`). Deliberately not
|
||||
/// accepting shorthand/`#rgba` so the value is safe to drop straight into CSS.
|
||||
fn is_hex_color(s: &str) -> bool {
|
||||
let bytes = s.as_bytes();
|
||||
bytes.len() == 7 && bytes[0] == b'#' && bytes[1..].iter().all(|b| b.is_ascii_hexdigit())
|
||||
}
|
||||
|
||||
pub async fn get_export_jobs(
|
||||
State(state): State<AppState>,
|
||||
RequireAdmin(_auth): RequireAdmin,
|
||||
@@ -172,61 +324,106 @@ pub async fn get_export_jobs(
|
||||
|
||||
// ── Export download endpoints (authenticated guests) ─────────────────────────
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct DownloadQuery {
|
||||
pub ticket: String,
|
||||
}
|
||||
|
||||
/// Mint a short-lived ticket for a browser-driven export download. The download
|
||||
/// is a top-level navigation so the multi-GB ZIP streams straight to disk instead
|
||||
/// of being buffered in memory by `fetch()` + `blob()` — but a navigation can't
|
||||
/// carry an `Authorization` header, so the client exchanges its Bearer token for
|
||||
/// an opaque ticket here, then hits `/export/zip?ticket=...`. Reuses the same
|
||||
/// single-use, 30s-TTL store as the SSE stream.
|
||||
pub async fn export_ticket(
|
||||
State(state): State<AppState>,
|
||||
auth: crate::auth::middleware::AuthUser,
|
||||
) -> Json<serde_json::Value> {
|
||||
// NOTE: intentionally NOT gated on `is_banned`. A banned user keeps *read* access
|
||||
// by design (USER_JOURNEYS §10.3, FEATURES: "Can still download the export once
|
||||
// released — Spec design choice"). The export is read-only, so it stays available
|
||||
// to them, consistent with the read-only-ban model.
|
||||
let ticket = state.sse_tickets.issue(auth.token_hash);
|
||||
Json(serde_json::json!({ "ticket": ticket }))
|
||||
}
|
||||
|
||||
/// Validate a download ticket (single-use) and confirm its session still exists.
|
||||
/// Resolve a single-use download ticket to the user who minted it. The caller needs the
|
||||
/// id to key the export rate limit per-user (see `enforce_export_rate`).
|
||||
async fn authenticate_download_ticket(state: &AppState, ticket: &str) -> Result<Uuid, AppError> {
|
||||
let token_hash = state
|
||||
.sse_tickets
|
||||
.consume(ticket)
|
||||
.ok_or_else(|| AppError::Unauthorized("Ticket ungültig oder abgelaufen.".into()))?;
|
||||
let session = crate::models::session::Session::find_by_token_hash(&state.pool, &token_hash)
|
||||
.await
|
||||
.map_err(|e| AppError::Internal(e.into()))?
|
||||
.ok_or_else(|| AppError::Unauthorized("Sitzung nicht gefunden.".into()))?;
|
||||
Ok(session.user_id)
|
||||
}
|
||||
|
||||
pub async fn download_zip(
|
||||
State(state): State<AppState>,
|
||||
_auth: crate::auth::middleware::AuthUser,
|
||||
headers: HeaderMap,
|
||||
Query(q): Query<DownloadQuery>,
|
||||
) -> Result<axum::response::Response, AppError> {
|
||||
let ip = client_ip(&headers, "unknown");
|
||||
let limit = get_config_usize(&state.pool, "export_rate_per_day", 3).await;
|
||||
if !state.rate_limiter.check(format!("export:{ip}"), limit, Duration::from_secs(86400)) {
|
||||
return Err(AppError::TooManyRequests("Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(), None));
|
||||
}
|
||||
let user_id = authenticate_download_ticket(&state, &q.ticket).await?;
|
||||
enforce_export_rate(&state, user_id).await?;
|
||||
|
||||
let event = crate::models::event::Event::find_by_slug(&state.pool, &state.config.event_slug)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
let path =
|
||||
resolve_export_file(&state, "zip", "Der ZIP-Export ist noch nicht verfügbar.").await?;
|
||||
serve_file(path, "Gallery.zip", "application/zip").await
|
||||
}
|
||||
|
||||
if !event.export_zip_ready {
|
||||
return Err(AppError::NotFound(
|
||||
"Der ZIP-Export ist noch nicht verfügbar.".into(),
|
||||
));
|
||||
}
|
||||
/// Resolve the on-disk path of the CURRENT export generation — readiness check and path lookup in
|
||||
/// ONE read, through the `export_current` view (migration 014).
|
||||
///
|
||||
/// This used to be two statements: the caller checked `event.export_zip_ready`, then this fetched
|
||||
/// `export_job.file_path`. A reopen landing between them served a keepsake that should have 404'd —
|
||||
/// the same stale-keepsake class, leaking through the read path, and it existed only because
|
||||
/// readiness was a stored copy rather than a derivation. `export_current` gives both facts from one
|
||||
/// consistent snapshot: it yields a row only when the event is released AND the job is `done` at the
|
||||
/// event's current epoch, so "is it ready" and "which file" can no longer disagree.
|
||||
async fn resolve_export_file(
|
||||
state: &AppState,
|
||||
export_type: &str,
|
||||
not_ready_msg: &str,
|
||||
) -> Result<std::path::PathBuf, AppError> {
|
||||
let file_path: Option<(Option<String>,)> = sqlx::query_as(
|
||||
"SELECT c.file_path FROM export_current c
|
||||
JOIN event e ON e.id = c.event_id
|
||||
WHERE e.slug = $1 AND c.type = $2::export_type AND c.status = 'done'",
|
||||
)
|
||||
.bind(&state.config.event_slug)
|
||||
.bind(export_type)
|
||||
.fetch_optional(&state.pool)
|
||||
.await
|
||||
.map_err(|e| AppError::Internal(e.into()))?;
|
||||
|
||||
let path = state.config.media_path.join("exports").join("Gallery.zip");
|
||||
let Some((Some(rel),)) = file_path else {
|
||||
return Err(AppError::NotFound(not_ready_msg.into()));
|
||||
};
|
||||
|
||||
// `file_path` is stored as `exports/<name>`; the base dir is already `export_path`, so
|
||||
// join only the file name (defends against any absolute/`..` content too).
|
||||
let name = std::path::Path::new(&rel)
|
||||
.file_name()
|
||||
.ok_or_else(|| AppError::NotFound("Exportdatei nicht gefunden.".into()))?;
|
||||
let path = state.config.export_path.join(name);
|
||||
if !path.exists() {
|
||||
return Err(AppError::NotFound("Exportdatei nicht gefunden.".into()));
|
||||
}
|
||||
|
||||
serve_file(path, "Gallery.zip", "application/zip").await
|
||||
Ok(path)
|
||||
}
|
||||
|
||||
pub async fn download_html(
|
||||
State(state): State<AppState>,
|
||||
_auth: crate::auth::middleware::AuthUser,
|
||||
headers: HeaderMap,
|
||||
Query(q): Query<DownloadQuery>,
|
||||
) -> Result<axum::response::Response, AppError> {
|
||||
let ip = client_ip(&headers, "unknown");
|
||||
let limit = get_config_usize(&state.pool, "export_rate_per_day", 3).await;
|
||||
if !state.rate_limiter.check(format!("export:{ip}"), limit, Duration::from_secs(86400)) {
|
||||
return Err(AppError::TooManyRequests("Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(), None));
|
||||
}
|
||||
|
||||
let event = crate::models::event::Event::find_by_slug(&state.pool, &state.config.event_slug)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
|
||||
if !event.export_html_ready {
|
||||
return Err(AppError::NotFound(
|
||||
"Der HTML-Export ist noch nicht verfügbar.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
let path = state.config.media_path.join("exports").join("Memories.zip");
|
||||
if !path.exists() {
|
||||
return Err(AppError::NotFound("Exportdatei nicht gefunden.".into()));
|
||||
}
|
||||
let user_id = authenticate_download_ticket(&state, &q.ticket).await?;
|
||||
enforce_export_rate(&state, user_id).await?;
|
||||
|
||||
let path =
|
||||
resolve_export_file(&state, "html", "Der HTML-Export ist noch nicht verfügbar.").await?;
|
||||
serve_file(path, "Memories.zip", "application/zip").await
|
||||
}
|
||||
|
||||
@@ -236,7 +433,7 @@ async fn serve_file(
|
||||
content_type: &str,
|
||||
) -> Result<axum::response::Response, AppError> {
|
||||
use axum::body::Body;
|
||||
use axum::http::{header, Response, StatusCode};
|
||||
use axum::http::{Response, StatusCode, header};
|
||||
use tokio_util::io::ReaderStream;
|
||||
|
||||
let file = tokio::fs::File::open(&path)
|
||||
@@ -272,8 +469,24 @@ pub async fn export_status(
|
||||
|
||||
let released = event.export_released_at.is_some();
|
||||
|
||||
let jobs: Vec<(String, String, i16)> = sqlx::query_as(
|
||||
"SELECT type::text, status::text, progress_pct FROM export_job WHERE event_id = $1",
|
||||
// ONE statement: the epoch comparison happens inside the query, against a single snapshot.
|
||||
// Binding the epoch read by a previous statement would let a release/regeneration commit in
|
||||
// between and yield `{released: true, zip: locked, html: locked}` — a state that never existed.
|
||||
//
|
||||
// Only jobs at the CURRENT epoch are reported. A row left behind by a retired generation (a
|
||||
// worker superseded mid-run) is meaningless — surfacing its frozen `running`/77% would show a
|
||||
// progress bar that never moves for a keepsake nobody is building. It reads as "locked" (no
|
||||
// current job), which is exactly what it is.
|
||||
// `error_message` is carried here, not just on the admin dashboard's job list. The host is the
|
||||
// one who releases the keepsake and the one who owns the "Erneut versuchen" button, but this
|
||||
// endpoint used to hand them a bare `failed` — so a fully actionable reason (notably the disk
|
||||
// preflight's "needs X GB, Y GB free") was written to the row and then shown to nobody who
|
||||
// could act on it. An admin-only diagnostic is not a diagnostic for the person on the spot.
|
||||
let jobs: Vec<(String, String, i16, Option<String>)> = sqlx::query_as(
|
||||
"SELECT j.type::text, j.status::text, j.progress_pct, j.error_message
|
||||
FROM export_job j
|
||||
JOIN event e ON e.id = j.event_id
|
||||
WHERE e.id = $1 AND j.epoch = e.export_epoch",
|
||||
)
|
||||
.bind(event.id)
|
||||
.fetch_all(&state.pool)
|
||||
@@ -281,11 +494,21 @@ pub async fn export_status(
|
||||
|
||||
let job_status = |type_name: &str| {
|
||||
jobs.iter()
|
||||
.find(|(t, _, _)| t == type_name)
|
||||
.map(|(_, status, pct)| {
|
||||
serde_json::json!({ "status": status, "progress_pct": pct })
|
||||
.find(|(t, _, _, _)| t == type_name)
|
||||
.map(|(_, status, pct, err)| {
|
||||
serde_json::json!({
|
||||
"status": status,
|
||||
"progress_pct": pct,
|
||||
// Only on a failure. A stale message left on a row that has since been re-armed
|
||||
// would otherwise show an error next to a running progress bar.
|
||||
"error_message": if status == "failed" { err.clone() } else { None },
|
||||
})
|
||||
})
|
||||
.unwrap_or_else(|| {
|
||||
serde_json::json!({
|
||||
"status": "locked", "progress_pct": 0, "error_message": null,
|
||||
})
|
||||
})
|
||||
.unwrap_or_else(|| serde_json::json!({ "status": "locked", "progress_pct": 0 }))
|
||||
};
|
||||
|
||||
Ok(Json(serde_json::json!({
|
||||
@@ -295,12 +518,28 @@ pub async fn export_status(
|
||||
})))
|
||||
}
|
||||
|
||||
async fn get_config_usize(pool: &sqlx::PgPool, key: &str, default: usize) -> usize {
|
||||
let row: Option<(String,)> =
|
||||
sqlx::query_as("SELECT value FROM config WHERE key = $1")
|
||||
.bind(key)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
.unwrap_or(None);
|
||||
row.and_then(|r| r.0.parse().ok()).unwrap_or(default)
|
||||
/// Centralised guard for the export rate limit. Same pattern as upload/feed: master
|
||||
/// switch + per-endpoint switch + numeric value, all stored in `config` and read on
|
||||
/// each request.
|
||||
async fn enforce_export_rate(state: &AppState, user_id: Uuid) -> Result<(), AppError> {
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let export_rate_on = config::get_bool(&state.config_cache, "export_rate_enabled", true).await;
|
||||
if !(rate_limits_on && export_rate_on) {
|
||||
return Ok(());
|
||||
}
|
||||
let limit = config::get_usize(&state.config_cache, "export_rate_per_day", 3).await;
|
||||
// Keyed per-user. This was the worst of the IP-keyed limiters: 3 downloads per DAY
|
||||
// shared across every guest behind the venue's public IP, so the fourth person to
|
||||
// fetch their keepsake was locked out until the next day.
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("export:{user_id}"),
|
||||
limit,
|
||||
Duration::from_secs(86400),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1,15 +1,14 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use axum::extract::{Query, State};
|
||||
use axum::http::HeaderMap;
|
||||
use axum::Json;
|
||||
use axum::extract::{Query, State};
|
||||
use chrono::{DateTime, Utc};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::auth::middleware::AuthUser;
|
||||
use crate::error::AppError;
|
||||
use crate::services::rate_limiter::client_ip;
|
||||
use crate::services::config;
|
||||
use crate::state::AppState;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
@@ -26,6 +25,8 @@ pub struct FeedUpload {
|
||||
pub uploader_name: String,
|
||||
pub preview_url: Option<String>,
|
||||
pub thumbnail_url: Option<String>,
|
||||
/// Big-screen (~2048px) variant for the diashow. Absent until the derivative exists.
|
||||
pub display_url: Option<String>,
|
||||
pub mime_type: String,
|
||||
pub caption: Option<String>,
|
||||
pub like_count: i64,
|
||||
@@ -47,6 +48,7 @@ struct FeedRow {
|
||||
uploader_name: String,
|
||||
preview_path: Option<String>,
|
||||
thumbnail_path: Option<String>,
|
||||
display_path: Option<String>,
|
||||
mime_type: String,
|
||||
caption: Option<String>,
|
||||
like_count: i64,
|
||||
@@ -57,60 +59,74 @@ struct FeedRow {
|
||||
pub async fn feed(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
headers: HeaderMap,
|
||||
Query(q): Query<FeedQuery>,
|
||||
) -> Result<Json<FeedResponse>, AppError> {
|
||||
let ip = client_ip(&headers, "unknown");
|
||||
let rate_limit = get_config_usize(&state.pool, "feed_rate_per_min", 60).await;
|
||||
if !state.rate_limiter.check(format!("feed:{ip}"), rate_limit, Duration::from_secs(60)) {
|
||||
return Err(AppError::TooManyRequests("Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(), None));
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let feed_rate_on = config::get_bool(&state.config_cache, "feed_rate_enabled", true).await;
|
||||
if rate_limits_on && feed_rate_on {
|
||||
let rate_limit = config::get_usize(&state.config_cache, "feed_rate_per_min", 60).await;
|
||||
// Keyed per-user, exactly like `feed_delta` below: at a venue every guest shares
|
||||
// one public IP, so an IP key gave the whole party a single 60/min bucket and the
|
||||
// fastest scroller starved everyone else.
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("feed:{}", auth.user_id),
|
||||
rate_limit,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let limit = q.limit.unwrap_or(20).min(100);
|
||||
|
||||
// Resolve the cursor to a (created_at, id) position. The pair is compared as a
|
||||
// tuple so ties on created_at break on id — keyset pagination on created_at
|
||||
// alone would silently drop rows sharing a timestamp across a page boundary.
|
||||
let (cursor_time, cursor_id) = match q.cursor {
|
||||
Some(c) => match get_cursor_pos(&state.pool, c).await {
|
||||
Some((t, id)) => (Some(t), Some(id)),
|
||||
None => (None, None),
|
||||
},
|
||||
None => (None, None),
|
||||
};
|
||||
|
||||
let rows = if let Some(hashtag) = &q.hashtag {
|
||||
let tag = hashtag.trim().trim_start_matches('#').to_lowercase();
|
||||
sqlx::query_as::<_, FeedRow>(
|
||||
"SELECT v.id, v.user_id, v.uploader_name, v.preview_path, v.thumbnail_path,
|
||||
v.mime_type, v.caption, v.like_count, v.comment_count, v.created_at
|
||||
v.display_path, v.mime_type, v.caption, v.like_count, v.comment_count,
|
||||
v.created_at
|
||||
FROM v_feed v
|
||||
JOIN upload_hashtag uh ON uh.upload_id = v.id
|
||||
JOIN hashtag h ON h.id = uh.hashtag_id AND h.tag = $1
|
||||
WHERE v.event_id = $2
|
||||
AND ($3::timestamptz IS NULL OR v.created_at < $3)
|
||||
ORDER BY v.created_at DESC
|
||||
LIMIT $4",
|
||||
AND ($3::timestamptz IS NULL OR (v.created_at, v.id) < ($3, $4))
|
||||
ORDER BY v.created_at DESC, v.id DESC
|
||||
LIMIT $5",
|
||||
)
|
||||
.bind(&tag)
|
||||
.bind(auth.event_id)
|
||||
.bind(
|
||||
if let Some(cursor) = q.cursor {
|
||||
get_cursor_time(&state.pool, cursor).await
|
||||
} else {
|
||||
None
|
||||
},
|
||||
)
|
||||
.bind(cursor_time)
|
||||
.bind(cursor_id)
|
||||
.bind(limit + 1)
|
||||
.fetch_all(&state.pool)
|
||||
.await?
|
||||
} else {
|
||||
sqlx::query_as::<_, FeedRow>(
|
||||
"SELECT id, user_id, uploader_name, preview_path, thumbnail_path,
|
||||
mime_type, caption, like_count, comment_count, created_at
|
||||
display_path, mime_type, caption, like_count, comment_count, created_at
|
||||
FROM v_feed
|
||||
WHERE event_id = $1
|
||||
AND ($2::timestamptz IS NULL OR created_at < $2)
|
||||
ORDER BY created_at DESC
|
||||
LIMIT $3",
|
||||
AND ($2::timestamptz IS NULL OR (created_at, id) < ($2, $3))
|
||||
ORDER BY created_at DESC, id DESC
|
||||
LIMIT $4",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.bind(
|
||||
if let Some(cursor) = q.cursor {
|
||||
get_cursor_time(&state.pool, cursor).await
|
||||
} else {
|
||||
None
|
||||
},
|
||||
)
|
||||
.bind(cursor_time)
|
||||
.bind(cursor_id)
|
||||
.bind(limit + 1)
|
||||
.fetch_all(&state.pool)
|
||||
.await?
|
||||
@@ -118,7 +134,11 @@ pub async fn feed(
|
||||
|
||||
let has_more = rows.len() as i64 > limit;
|
||||
let rows: Vec<FeedRow> = rows.into_iter().take(limit as usize).collect();
|
||||
let next_cursor = if has_more { rows.last().map(|r| r.id) } else { None };
|
||||
let next_cursor = if has_more {
|
||||
rows.last().map(|r| r.id)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Batch check which uploads the current user has liked
|
||||
let upload_ids: Vec<Uuid> = rows.iter().map(|r| r.id).collect();
|
||||
@@ -127,8 +147,21 @@ pub async fn feed(
|
||||
let uploads = rows
|
||||
.into_iter()
|
||||
.map(|r| {
|
||||
let preview_url = r.preview_path.map(|p| format!("/media/{p}"));
|
||||
let thumbnail_url = r.thumbnail_path.map(|p| format!("/media/{p}"));
|
||||
// Gated media aliases (visibility-checked, direct /media blocked). Emit the
|
||||
// URL only when the variant actually exists — the URL is what signals the
|
||||
// client which variant to load.
|
||||
let preview_url = r
|
||||
.preview_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/preview", r.id));
|
||||
let thumbnail_url = r
|
||||
.thumbnail_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/thumbnail", r.id));
|
||||
let display_url = r
|
||||
.display_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/display", r.id));
|
||||
FeedUpload {
|
||||
liked_by_me: liked_set.contains(&r.id),
|
||||
id: r.id,
|
||||
@@ -136,6 +169,7 @@ pub async fn feed(
|
||||
uploader_name: r.uploader_name,
|
||||
preview_url,
|
||||
thumbnail_url,
|
||||
display_url,
|
||||
mime_type: r.mime_type,
|
||||
caption: r.caption,
|
||||
like_count: r.like_count,
|
||||
@@ -160,6 +194,22 @@ pub struct DeltaQuery {
|
||||
pub struct DeltaResponse {
|
||||
pub uploads: Vec<FeedUpload>,
|
||||
pub deleted_ids: Vec<Uuid>,
|
||||
/// Users whose uploads became hidden (banned / uploads_hidden) since `since`. A ban is
|
||||
/// not a soft-delete, so it never appears in `deleted_ids`; without this, a client that
|
||||
/// missed the ephemeral `user-hidden` SSE (a reconnecting projector, most acutely) would
|
||||
/// keep displaying the banned user's already-loaded slides. The client evicts every
|
||||
/// upload from these users on receipt.
|
||||
pub hidden_user_ids: Vec<Uuid>,
|
||||
/// True when the upload query hit `DELTA_LIMIT`: the response carries only the
|
||||
/// newest slice of the gap, so the client must fall back to a full feed refresh
|
||||
/// rather than merging (the older missed uploads are absent and unrecoverable
|
||||
/// via a later delta, which advances `since` past them).
|
||||
pub truncated: bool,
|
||||
/// The server's clock at the moment this delta was computed. The client advances its
|
||||
/// reconnect cursor from THIS, never `new Date()` — a browser clock even seconds fast
|
||||
/// would otherwise silently skip uploads whose server `created_at` falls in the skew
|
||||
/// window (they'd never reappear without a hard refresh).
|
||||
pub server_time: DateTime<Utc>,
|
||||
}
|
||||
|
||||
pub async fn feed_delta(
|
||||
@@ -167,21 +217,78 @@ pub async fn feed_delta(
|
||||
auth: AuthUser,
|
||||
Query(q): Query<DeltaQuery>,
|
||||
) -> Result<Json<DeltaResponse>, AppError> {
|
||||
// Rate-limit the delta the same way as the paginated feed. Without this, ~100 clients
|
||||
// reconnecting at once (post-outage, or a flapping network) each fire an unbounded
|
||||
// delta fetch — a reconnect stampede. Keyed per-user so one client can't starve others
|
||||
// behind a shared NAT.
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let feed_rate_on = config::get_bool(&state.config_cache, "feed_rate_enabled", true).await;
|
||||
if rate_limits_on && feed_rate_on {
|
||||
let rate_limit = config::get_usize(&state.config_cache, "feed_rate_per_min", 60).await;
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("feed_delta:{}", auth.user_id),
|
||||
rate_limit,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Anfragen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Anchor the next cursor to the DB clock, captured *before* the queries so an upload
|
||||
// committed during this handler is re-fetched next time rather than skipped (a
|
||||
// duplicate id merges idempotently on the client; a miss is unrecoverable).
|
||||
let server_time: DateTime<Utc> = sqlx::query_scalar("SELECT NOW()")
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
|
||||
// Bounded like the paginated feed: a stale `since` could otherwise pull the
|
||||
// entire event's uploads in one response. If a client hits the cap it should
|
||||
// fall back to a full feed refresh rather than another delta.
|
||||
const DELTA_LIMIT: i64 = 200;
|
||||
// `>= since` (not `>`): `created_at` is not unique, so a strict `>` anchored to the exact
|
||||
// timestamp of the last-seen upload silently drops a SECOND upload committed in the same
|
||||
// microsecond — an unrecoverable live-update miss. `>=` re-includes the boundary rows;
|
||||
// the client merges by id (see the feed-delta handler's `seen` set), so the duplicate is
|
||||
// harmless while the tied upload is no longer lost. The cursor still advances to the
|
||||
// response's `server_time`, so this doesn't re-fetch on every subsequent delta.
|
||||
let rows = sqlx::query_as::<_, FeedRow>(
|
||||
"SELECT id, user_id, uploader_name, preview_path, thumbnail_path,
|
||||
mime_type, caption, like_count, comment_count, created_at
|
||||
display_path, mime_type, caption, like_count, comment_count, created_at
|
||||
FROM v_feed
|
||||
WHERE event_id = $1 AND created_at > $2
|
||||
ORDER BY created_at DESC",
|
||||
WHERE event_id = $1 AND created_at >= $2
|
||||
ORDER BY created_at DESC, id DESC
|
||||
LIMIT $3",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.bind(q.since)
|
||||
.bind(DELTA_LIMIT)
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
|
||||
// Hit the cap => this is only the newest slice of a larger gap. Signal the
|
||||
// client to full-refresh instead of merging a partial delta.
|
||||
let truncated = rows.len() as i64 >= DELTA_LIMIT;
|
||||
|
||||
// `>=` for the same tie-break reason as the uploads query above; re-signalling an
|
||||
// already-removed id is idempotent on the client (it filters its list by these ids).
|
||||
let deleted_ids: Vec<(Uuid,)> = sqlx::query_as(
|
||||
"SELECT id FROM upload
|
||||
WHERE event_id = $1 AND deleted_at IS NOT NULL AND deleted_at >= $2",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.bind(q.since)
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
|
||||
let deleted_ids: Vec<(Uuid,)> = sqlx::query_as(
|
||||
"SELECT id FROM upload
|
||||
WHERE event_id = $1 AND deleted_at IS NOT NULL AND deleted_at > $2",
|
||||
// Users hidden since the cursor (ban / uploads_hidden). Uncapped like `deleted_ids` and
|
||||
// for the same reason — an eviction the client misses is unrecoverable via a later delta.
|
||||
let hidden_user_ids: Vec<(Uuid,)> = sqlx::query_as(
|
||||
"SELECT id FROM \"user\"
|
||||
WHERE event_id = $1 AND uploads_hidden = TRUE
|
||||
AND uploads_hidden_at IS NOT NULL AND uploads_hidden_at >= $2",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.bind(q.since)
|
||||
@@ -198,8 +305,18 @@ pub async fn feed_delta(
|
||||
id: r.id,
|
||||
user_id: r.user_id,
|
||||
uploader_name: r.uploader_name,
|
||||
preview_url: r.preview_path.map(|p| format!("/media/{p}")),
|
||||
thumbnail_url: r.thumbnail_path.map(|p| format!("/media/{p}")),
|
||||
preview_url: r
|
||||
.preview_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/preview", r.id)),
|
||||
thumbnail_url: r
|
||||
.thumbnail_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/thumbnail", r.id)),
|
||||
display_url: r
|
||||
.display_path
|
||||
.as_ref()
|
||||
.map(|_| format!("/api/v1/upload/{}/display", r.id)),
|
||||
mime_type: r.mime_type,
|
||||
caption: r.caption,
|
||||
like_count: r.like_count,
|
||||
@@ -211,6 +328,9 @@ pub async fn feed_delta(
|
||||
Ok(Json(DeltaResponse {
|
||||
uploads,
|
||||
deleted_ids: deleted_ids.into_iter().map(|r| r.0).collect(),
|
||||
hidden_user_ids: hidden_user_ids.into_iter().map(|r| r.0).collect(),
|
||||
truncated,
|
||||
server_time,
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -224,12 +344,11 @@ pub async fn hashtags(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<Json<Vec<HashtagCount>>, AppError> {
|
||||
let rows: Vec<(String, i64)> = sqlx::query_as(
|
||||
"SELECT tag, upload_count FROM v_hashtag_counts WHERE event_id = $1",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
let rows: Vec<(String, i64)> =
|
||||
sqlx::query_as("SELECT tag, upload_count FROM v_hashtag_counts WHERE event_id = $1")
|
||||
.bind(auth.event_id)
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
|
||||
Ok(Json(
|
||||
rows.into_iter()
|
||||
@@ -238,24 +357,17 @@ pub async fn hashtags(
|
||||
))
|
||||
}
|
||||
|
||||
async fn get_config_usize(pool: &sqlx::PgPool, key: &str, default: usize) -> usize {
|
||||
let row: Option<(String,)> =
|
||||
sqlx::query_as("SELECT value FROM config WHERE key = $1")
|
||||
.bind(key)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
.unwrap_or(None);
|
||||
row.and_then(|r| r.0.parse().ok()).unwrap_or(default)
|
||||
}
|
||||
|
||||
async fn get_cursor_time(pool: &sqlx::PgPool, cursor_id: Uuid) -> Option<DateTime<Utc>> {
|
||||
let row: Option<(DateTime<Utc>,)> =
|
||||
sqlx::query_as("SELECT created_at FROM upload WHERE id = $1")
|
||||
/// Resolve a cursor id to its `(created_at, id)` position. Both are needed:
|
||||
/// `created_at` alone isn't unique, so pagination must break ties on `id` to
|
||||
/// avoid silently dropping rows that share a timestamp across a page boundary.
|
||||
async fn get_cursor_pos(pool: &sqlx::PgPool, cursor_id: Uuid) -> Option<(DateTime<Utc>, Uuid)> {
|
||||
let row: Option<(DateTime<Utc>, Uuid)> =
|
||||
sqlx::query_as("SELECT created_at, id FROM upload WHERE id = $1")
|
||||
.bind(cursor_id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
.ok()?;
|
||||
row.map(|r| r.0)
|
||||
row
|
||||
}
|
||||
|
||||
async fn get_liked_set(
|
||||
@@ -266,14 +378,13 @@ async fn get_liked_set(
|
||||
if upload_ids.is_empty() {
|
||||
return std::collections::HashSet::new();
|
||||
}
|
||||
let rows: Vec<(Uuid,)> = sqlx::query_as(
|
||||
"SELECT upload_id FROM \"like\" WHERE user_id = $1 AND upload_id = ANY($2)",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(upload_ids)
|
||||
.fetch_all(pool)
|
||||
.await
|
||||
.unwrap_or_default();
|
||||
let rows: Vec<(Uuid,)> =
|
||||
sqlx::query_as("SELECT upload_id FROM \"like\" WHERE user_id = $1 AND upload_id = ANY($2)")
|
||||
.bind(user_id)
|
||||
.bind(upload_ids)
|
||||
.fetch_all(pool)
|
||||
.await
|
||||
.unwrap_or_default();
|
||||
|
||||
rows.into_iter().map(|r| r.0).collect()
|
||||
}
|
||||
|
||||
43
backend/src/handlers/health.rs
Normal file
43
backend/src/handlers/health.rs
Normal file
@@ -0,0 +1,43 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use axum::extract::State;
|
||||
use axum::http::StatusCode;
|
||||
|
||||
use crate::state::AppState;
|
||||
|
||||
/// How long a readiness probe waits for the database before calling the app not-ready.
|
||||
///
|
||||
/// Deliberately shorter than the pool's own `acquire_timeout`: a saturated pool IS a
|
||||
/// not-ready condition, and reporting it as such is the point. Do not "fix" the two to
|
||||
/// match — that would make the probe wait out the very saturation it exists to surface.
|
||||
const READY_TIMEOUT: Duration = Duration::from_secs(2);
|
||||
|
||||
/// Readiness: can this instance actually serve a request end to end?
|
||||
///
|
||||
/// Separate from `/health` (liveness) ON PURPOSE, and the compose healthcheck must keep
|
||||
/// pointing at `/health`. `caddy` declares `depends_on: app: {condition: service_healthy}`,
|
||||
/// so a DB-dependent healthcheck would turn a transient Postgres hiccup during boot into
|
||||
/// the reverse proxy refusing to start — converting a blip into a total outage. This route
|
||||
/// is for an external monitor, which should page rather than restart.
|
||||
///
|
||||
/// The gap this closes: every request path touches the database, so with a dead or
|
||||
/// saturated pool the app is useless while `/health` still answers "ok" — the disk-full
|
||||
/// endgame (media volume fills, Postgres cannot write WAL) stayed green all the way down.
|
||||
pub async fn ready(State(state): State<AppState>) -> (StatusCode, &'static str) {
|
||||
let probe = sqlx::query_scalar::<_, i32>("SELECT 1").fetch_one(&state.pool);
|
||||
|
||||
match tokio::time::timeout(READY_TIMEOUT, probe).await {
|
||||
Ok(Ok(_)) => (StatusCode::OK, "ready"),
|
||||
Ok(Err(e)) => {
|
||||
tracing::warn!(error = ?e, "readiness probe failed: database unreachable");
|
||||
(StatusCode::SERVICE_UNAVAILABLE, "database unavailable")
|
||||
}
|
||||
Err(_) => {
|
||||
tracing::warn!(
|
||||
timeout_secs = READY_TIMEOUT.as_secs(),
|
||||
"readiness probe timed out; the pool is saturated or the database is hung"
|
||||
);
|
||||
(StatusCode::SERVICE_UNAVAILABLE, "database timeout")
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
use axum::Json;
|
||||
use axum::extract::{Path, State};
|
||||
use axum::http::StatusCode;
|
||||
use axum::Json;
|
||||
use chrono::{DateTime, Utc};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use uuid::Uuid;
|
||||
@@ -9,8 +9,11 @@ use crate::auth::middleware::RequireHost;
|
||||
use crate::error::AppError;
|
||||
use crate::models::comment::Comment;
|
||||
use crate::models::event::Event;
|
||||
use crate::models::session::Session;
|
||||
use crate::models::upload::Upload;
|
||||
use crate::state::AppState;
|
||||
use crate::models::user::UserRole;
|
||||
use crate::services::export::Affects;
|
||||
use crate::state::{AppState, SseEvent};
|
||||
|
||||
// ── DTOs ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -32,11 +35,52 @@ pub struct EventStatus {
|
||||
pub is_active: bool,
|
||||
pub uploads_locked: bool,
|
||||
pub export_released: bool,
|
||||
/// Free space on the volume the keepsake is written to. `None` when the mount can't be
|
||||
/// resolved — the UI hides the widget rather than rendering a confident zero.
|
||||
pub disk_free_bytes: Option<u64>,
|
||||
/// What a full keepsake build would need right now (both halves).
|
||||
pub keepsake_required_bytes: u64,
|
||||
/// Whether the host should be warned. See [`disk_is_low`].
|
||||
pub disk_low: bool,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct BanRequest {
|
||||
pub hide_uploads: bool,
|
||||
/// Absolute floor below which free space is worth surfacing regardless of gallery size — the
|
||||
/// threshold the README has carried on the roadmap since v1.
|
||||
const LOW_DISK_FLOOR_BYTES: u64 = 10_000_000_000;
|
||||
|
||||
/// Is free space low enough that the host needs to know?
|
||||
///
|
||||
/// Two triggers, because a fixed threshold answers the wrong question. `postgres_data`,
|
||||
/// `media_data` and `exports_data` are all Docker named volumes on one filesystem, so a full disk
|
||||
/// does not degrade one subsystem — it stops Postgres writing and takes the event down. That is
|
||||
/// what the absolute floor is for.
|
||||
///
|
||||
/// The second trigger is the one that actually earns its place: the keepsake needs room for two
|
||||
/// gallery-sized archives, and the only moment a host can do anything about that is BEFORE they
|
||||
/// release. Warning at "you could not build the keepsake right now" turns a post-event dead end
|
||||
/// into a decision someone can still make.
|
||||
fn disk_is_low(free: u64, keepsake_required: u64) -> bool {
|
||||
free < LOW_DISK_FLOOR_BYTES || free < keepsake_required
|
||||
}
|
||||
|
||||
/// Count non-banned hosts/admins in the event OTHER than `excluding` — the operators
|
||||
/// who would remain if `excluding` were demoted or banned. Used to enforce the "an event
|
||||
/// always keeps at least one operator" floor.
|
||||
async fn remaining_operators(
|
||||
state: &AppState,
|
||||
event_id: Uuid,
|
||||
excluding: Uuid,
|
||||
) -> Result<i64, AppError> {
|
||||
let count = sqlx::query_scalar::<_, i64>(
|
||||
"SELECT COUNT(*) FROM \"user\"
|
||||
WHERE event_id = $1 AND id != $2
|
||||
AND role IN ('host', 'admin') AND is_banned = FALSE",
|
||||
)
|
||||
.bind(event_id)
|
||||
.bind(excluding)
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
Ok(count)
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
@@ -54,11 +98,29 @@ pub async fn get_event_status(
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
|
||||
// Measured on the EXPORT volume, not the media one: that is where the cliff is, and it is a
|
||||
// distinct mount point even when both are backed by the same filesystem. The cached reading is
|
||||
// right here — this is advisory, polled on every dashboard load, and a 15s-stale number costs
|
||||
// nothing (unlike the export preflight, which reads uncached because it is about to write).
|
||||
let free = state
|
||||
.disk_cache
|
||||
.snapshot(&state.config.export_path)
|
||||
.map(|d| d.free);
|
||||
let keepsake_required_bytes =
|
||||
crate::services::export::keepsake_space_required(&state.pool, event.id)
|
||||
.await
|
||||
.unwrap_or(0);
|
||||
|
||||
Ok(Json(EventStatus {
|
||||
name: event.name,
|
||||
is_active: event.is_active,
|
||||
uploads_locked: event.uploads_locked_at.is_some(),
|
||||
export_released: event.export_released_at.is_some(),
|
||||
disk_free_bytes: free,
|
||||
keepsake_required_bytes,
|
||||
// Unknown free space is NOT low. Fails open, exactly as the upload quota and the export
|
||||
// preflight do: a scary banner on an unreadable mount would train the host to ignore it.
|
||||
disk_low: free.is_some_and(|f| disk_is_low(f, keepsake_required_bytes)),
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -92,11 +154,13 @@ pub async fn ban_user(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(user_id): Path<Uuid>,
|
||||
Json(body): Json<BanRequest>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
// The ban request carries no body — ban always hides (no per-request options).
|
||||
// Cannot ban yourself or another host/admin
|
||||
if user_id == auth.user_id {
|
||||
return Err(AppError::BadRequest("Du kannst dich nicht selbst sperren.".into()));
|
||||
return Err(AppError::BadRequest(
|
||||
"Du kannst dich nicht selbst sperren.".into(),
|
||||
));
|
||||
}
|
||||
let target = sqlx::query_as::<_, (String,)>(
|
||||
"SELECT role::text FROM \"user\" WHERE id = $1 AND event_id = $2",
|
||||
@@ -107,30 +171,190 @@ pub async fn ban_user(
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
if target.0 == "admin" || (target.0 == "host" && auth.role != crate::models::user::UserRole::Admin) {
|
||||
return Err(AppError::Forbidden("Du kannst diesen Benutzer nicht sperren.".into()));
|
||||
if target.0 == "admin"
|
||||
|| (target.0 == "host" && auth.role != crate::models::user::UserRole::Admin)
|
||||
{
|
||||
return Err(AppError::Forbidden(
|
||||
"Du kannst diesen Benutzer nicht sperren.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// Floor: never leave the event with zero operators. Banning removes the target from
|
||||
// the active-operator pool, so refuse if they're the last non-banned host/admin.
|
||||
if target.0 == "host" && remaining_operators(&state, auth.event_id, user_id).await? == 0 {
|
||||
return Err(AppError::BadRequest(
|
||||
"Der letzte Host kann nicht gesperrt werden.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// Ban ALWAYS hides: a banned user's content is "gone" everywhere. The visibility
|
||||
// views/queries now also filter on `is_banned` (defense in depth), and we set
|
||||
// `uploads_hidden` so the existing `user-hidden` live-eviction path fires too. The old
|
||||
// opt-out checkbox is gone — `hide_uploads` in the request is ignored.
|
||||
//
|
||||
// We deliberately do NOT revoke the banned user's sessions. This is a *read-only ban*
|
||||
// by design (USER_JOURNEYS §10.3): the user keeps read access to the feed and can still
|
||||
// download the released export — writes and host/admin actions are what the ban blocks
|
||||
// (enforced live on the write handlers + Require{Host,Admin}). Revoking sessions would
|
||||
// contradict that model, break the documented "banned guest can still download the
|
||||
// keepsake" flow, and be ineffective anyway (the user could just /recover a new session).
|
||||
//
|
||||
// The ban and the keepsake invalidation are ONE transaction — see `host_delete_upload`.
|
||||
let mut tx = state.pool.begin().await?;
|
||||
sqlx::query(
|
||||
"UPDATE \"user\" SET is_banned = TRUE, uploads_hidden = $2 WHERE id = $1",
|
||||
"UPDATE \"user\"
|
||||
SET is_banned = TRUE, uploads_hidden = TRUE, uploads_hidden_at = NOW()
|
||||
WHERE id = $1 AND event_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(body.hide_uploads)
|
||||
.execute(&state.pool)
|
||||
.bind(auth.event_id)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
|
||||
// A ban hides the user's uploads EVERYWHERE — and the keepsake is the place that matters most,
|
||||
// because it is the copy people keep. The export already filters `is_banned = FALSE`, so a
|
||||
// FUTURE export excludes them; without this, an ALREADY-RELEASED archive would keep serving a
|
||||
// banned user's photos forever. Same class as a takedown, so same treatment.
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
Affects::Both,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
if let Some(r) = regen {
|
||||
start_regen(&state, r);
|
||||
}
|
||||
|
||||
// Evict their content live from every feed + the diashow so it disappears without
|
||||
// each viewer having to reload. (Their own SSE stream is separately dropped by the
|
||||
// is_banned revalidation in `sse::stream`, so they stop receiving live pushes while
|
||||
// retaining plain read access.)
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"user-hidden",
|
||||
serde_json::json!({ "user_id": user_id }).to_string(),
|
||||
));
|
||||
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
target_user_id = %user_id,
|
||||
event_id = %auth.event_id,
|
||||
"host: ban_user"
|
||||
);
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
pub async fn unban_user(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(user_id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
sqlx::query("UPDATE \"user\" SET is_banned = FALSE WHERE id = $1")
|
||||
.bind(user_id)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
// Mirror the ban guard: a host may only lift bans on guests, never on hosts or
|
||||
// admins. Without this a host could override an admin's ban of another host,
|
||||
// which is asymmetric with `ban_user` and lets a host escalate a peer back in.
|
||||
let target = sqlx::query_as::<_, (String,)>(
|
||||
"SELECT role::text FROM \"user\" WHERE id = $1 AND event_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(auth.event_id)
|
||||
.fetch_optional(&state.pool)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
if target.0 == "admin"
|
||||
|| (target.0 == "host" && auth.role != crate::models::user::UserRole::Admin)
|
||||
{
|
||||
return Err(AppError::Forbidden(
|
||||
"Du kannst diesen Benutzer nicht entsperren.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// Unban restores visibility too: ban set `uploads_hidden = TRUE`, so clearing only
|
||||
// `is_banned` would leave their content invisible. Clear all three (the timestamp too,
|
||||
// so a future ban stamps a fresh `uploads_hidden_at` the reconnect delta will replay).
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let result = sqlx::query(
|
||||
"UPDATE \"user\"
|
||||
SET is_banned = FALSE, uploads_hidden = FALSE, uploads_hidden_at = NULL
|
||||
WHERE id = $1 AND event_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(auth.event_id)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
if result.rows_affected() == 0 {
|
||||
return Err(AppError::NotFound("Benutzer nicht gefunden.".into()));
|
||||
}
|
||||
|
||||
// The mirror of the ban case: an unban RESTORES their uploads to the export query, so an
|
||||
// already-released keepsake is now missing content it should contain. Rebuild it.
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
Affects::Both,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
if let Some(r) = regen {
|
||||
start_regen(&state, r);
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
target_user_id = %user_id,
|
||||
event_id = %auth.event_id,
|
||||
"host: unban_user"
|
||||
);
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
/// Force a keepsake rebuild. The ESCAPE HATCH.
|
||||
///
|
||||
/// Without this, a failed or stranded export is terminal at runtime: `release_gallery` refuses an
|
||||
/// already-released event ("bereits freigegeben"), `recover_exports` only runs at boot, and there is
|
||||
/// no other retry path — so the host's only options were restarting the container or reopening the
|
||||
/// event (which unlocks uploads to every guest and discards the release). This is also the recovery
|
||||
/// path for a keepsake that went stale for any reason we haven't thought of.
|
||||
pub async fn rebuild_export(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(auth): RequireHost,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
Affects::Both,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
|
||||
let Some(r) = regen else {
|
||||
return Err(AppError::BadRequest(
|
||||
"Die Galerie ist nicht freigegeben — es gibt nichts neu zu erzeugen.".into(),
|
||||
));
|
||||
};
|
||||
|
||||
// No debounce: this is an explicit, deliberate host action, not a burst.
|
||||
for export_type in ["zip", "html"] {
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"export-progress",
|
||||
serde_json::json!({ "type": export_type, "progress_pct": 0 }).to_string(),
|
||||
));
|
||||
}
|
||||
crate::services::export::spawn_export_jobs(
|
||||
r.event_id,
|
||||
r.event_name,
|
||||
r.epoch,
|
||||
state.config.comments_enabled,
|
||||
std::time::Duration::ZERO,
|
||||
state.pool.clone(),
|
||||
state.config.media_path.clone(),
|
||||
state.config.export_path.clone(),
|
||||
state.sse_tx.clone(),
|
||||
);
|
||||
|
||||
tracing::info!(actor_user_id = %auth.user_id, "host: rebuild_export");
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
@@ -141,51 +365,324 @@ pub async fn set_role(
|
||||
Json(body): Json<SetRoleRequest>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
if user_id == auth.user_id {
|
||||
return Err(AppError::BadRequest("Du kannst deine eigene Rolle nicht ändern.".into()));
|
||||
return Err(AppError::BadRequest(
|
||||
"Du kannst deine eigene Rolle nicht ändern.".into(),
|
||||
));
|
||||
}
|
||||
let new_role = match body.role.as_str() {
|
||||
"guest" => "guest",
|
||||
"host" => "host",
|
||||
_ => return Err(AppError::BadRequest("Ungültige Rolle. Erlaubt: guest, host.".into())),
|
||||
_ => {
|
||||
return Err(AppError::BadRequest(
|
||||
"Ungültige Rolle. Erlaubt: guest, host.".into(),
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
// Look up the current role so we can apply the host-vs-admin guard. A plain host may
|
||||
// promote/demote GUESTS only; it may not change any host's or admin's role (see the
|
||||
// guard below — this closes the demote-a-peer-host→ban/PIN-reset takeover chain, F1).
|
||||
// Only an admin may change a host's role. Admins may do anything except change
|
||||
// themselves (blocked above).
|
||||
let target = sqlx::query_as::<_, (String,)>(
|
||||
"SELECT role::text FROM \"user\" WHERE id = $1 AND event_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(auth.event_id)
|
||||
.fetch_optional(&state.pool)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
// Admins are untouchable by hosts. A plain host also may not demote another
|
||||
// *host*: without this guard a host could demote a peer host to guest and then
|
||||
// ban / PIN-reset (→ account-takeover via /recover) them — the ban/pin-reset peer
|
||||
// guards key off the target's *current* role, so a prior demotion would launder
|
||||
// past them. Only an admin may change a host's role.
|
||||
if target.0 == "admin" || (target.0 == "host" && auth.role != UserRole::Admin) {
|
||||
return Err(AppError::Forbidden(
|
||||
"Du darfst die Rolle dieses Benutzers nicht ändern.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
// Floor: demoting the last non-banned host/admin to guest would leave the event with
|
||||
// no operator. Refuse.
|
||||
if new_role == "guest"
|
||||
&& target.0 == "host"
|
||||
&& remaining_operators(&state, auth.event_id, user_id).await? == 0
|
||||
{
|
||||
return Err(AppError::BadRequest(
|
||||
"Der letzte Host kann nicht zum Gast gemacht werden.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
sqlx::query("UPDATE \"user\" SET role = $2::user_role WHERE id = $1 AND event_id = $3")
|
||||
.bind(user_id)
|
||||
.bind(new_role)
|
||||
.bind(auth.event_id)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
target_user_id = %user_id,
|
||||
event_id = %auth.event_id,
|
||||
old_role = %target.0,
|
||||
new_role,
|
||||
"host: set_role"
|
||||
);
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct PinResetResponse {
|
||||
/// Plaintext PIN — shown to the operator **once**. Never persisted client-side.
|
||||
pub pin: String,
|
||||
}
|
||||
|
||||
/// Generate a fresh PIN for another user, returning the plaintext exactly once.
|
||||
///
|
||||
/// Authorisation:
|
||||
/// - Host caller → may reset **guest** PINs only.
|
||||
/// - Admin caller → may reset **guest** and **host** PINs (never another admin).
|
||||
/// - Target ≠ caller.
|
||||
pub async fn reset_user_pin(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(user_id): Path<Uuid>,
|
||||
) -> Result<Json<PinResetResponse>, AppError> {
|
||||
use rand::Rng;
|
||||
|
||||
if user_id == auth.user_id {
|
||||
return Err(AppError::BadRequest(
|
||||
"Du kannst deine eigene PIN nicht über diese Funktion zurücksetzen.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
let target = sqlx::query_as::<_, (String,)>(
|
||||
"SELECT role::text FROM \"user\" WHERE id = $1 AND event_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(auth.event_id)
|
||||
.fetch_optional(&state.pool)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
match (auth.role.clone(), target.0.as_str()) {
|
||||
(UserRole::Admin, "guest" | "host") => {}
|
||||
(UserRole::Host, "guest") => {}
|
||||
_ => {
|
||||
return Err(AppError::Forbidden(
|
||||
"Du darfst die PIN dieses Benutzers nicht zurücksetzen.".into(),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let pin: String = format!("{:04}", rand::rng().random_range(0..10000u32));
|
||||
let pin_hash = crate::auth::handlers::hash_password(pin.clone(), 12).await?;
|
||||
|
||||
sqlx::query(
|
||||
"UPDATE \"user\"
|
||||
SET recovery_pin_hash = $1,
|
||||
failed_pin_attempts = 0,
|
||||
pin_locked_until = NULL
|
||||
WHERE id = $2 AND event_id = $3",
|
||||
)
|
||||
.bind(&pin_hash)
|
||||
.bind(user_id)
|
||||
.bind(auth.event_id)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
// A PIN reset means the old credential is compromised/forgotten — revoke every
|
||||
// existing session so old devices must re-authenticate with the new PIN. This is a
|
||||
// security-relevant revoke: if it fails, the old sessions stay valid (sessions are
|
||||
// token- not PIN-bound), so surface the error in logs rather than swallowing it
|
||||
// silently while reporting success to the host.
|
||||
if let Err(e) = Session::delete_all_for_user(&state.pool, user_id).await {
|
||||
tracing::error!(error = ?e, user_id = %user_id, "PIN reset: failed to revoke sessions");
|
||||
}
|
||||
|
||||
// Resolve any pending in-app "I forgot my PIN" request for this user.
|
||||
let _ = sqlx::query("DELETE FROM pin_reset_request WHERE user_id = $1")
|
||||
.bind(user_id)
|
||||
.execute(&state.pool)
|
||||
.await;
|
||||
|
||||
// Notify the *recipient* device(s) if they happen to be online so they can clear
|
||||
// their cached local PIN. They'll save the new one on the next /recover.
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"pin-reset",
|
||||
serde_json::json!({ "user_id": user_id }).to_string(),
|
||||
));
|
||||
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
target_user_id = %user_id,
|
||||
event_id = %auth.event_id,
|
||||
"host: reset_user_pin"
|
||||
);
|
||||
|
||||
Ok(Json(PinResetResponse { pin }))
|
||||
}
|
||||
|
||||
#[derive(Serialize, sqlx::FromRow)]
|
||||
pub struct PinResetRequestSummary {
|
||||
pub id: Uuid,
|
||||
pub user_id: Uuid,
|
||||
pub display_name: String,
|
||||
pub created_at: DateTime<Utc>,
|
||||
}
|
||||
|
||||
/// List pending in-app PIN-reset requests so a host can action them (via the existing
|
||||
/// `reset_user_pin`, which also clears the request).
|
||||
pub async fn list_pin_reset_requests(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(auth): RequireHost,
|
||||
) -> Result<Json<Vec<PinResetRequestSummary>>, AppError> {
|
||||
let rows = sqlx::query_as::<_, PinResetRequestSummary>(
|
||||
"SELECT r.id, r.user_id, u.display_name, r.created_at
|
||||
FROM pin_reset_request r
|
||||
JOIN \"user\" u ON u.id = r.user_id
|
||||
WHERE r.event_id = $1
|
||||
ORDER BY r.created_at ASC",
|
||||
)
|
||||
.bind(auth.event_id)
|
||||
.fetch_all(&state.pool)
|
||||
.await?;
|
||||
Ok(Json(rows))
|
||||
}
|
||||
|
||||
/// Dismiss a PIN-reset request without resetting (e.g. the host couldn't verify the
|
||||
/// requester's identity).
|
||||
pub async fn dismiss_pin_reset_request(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
sqlx::query("DELETE FROM pin_reset_request WHERE id = $1 AND event_id = $2")
|
||||
.bind(id)
|
||||
.bind(auth.event_id)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
/// Content changed AFTER the gallery was released — regenerate the keepsake.
|
||||
///
|
||||
/// Deleting a photo used to remove it from the live feed but leave it in the already-generated
|
||||
/// archive FOREVER: the old `if ready { continue }` skip guaranteed the export was never rebuilt.
|
||||
/// For a takedown ("please remove my photo") that is the one place it most needs to disappear.
|
||||
///
|
||||
/// Bumping the epoch (while staying released) retires the current generation, so the stale archive
|
||||
/// stops being downloadable the instant the delete commits, and a fresh worker rebuilds it without
|
||||
/// the removed content. The download 404s in the meantime, which is the correct answer — serving
|
||||
/// the old archive would serve the deleted photo.
|
||||
/// Start the workers for a regeneration that was armed inside a (now-committed) transaction, and
|
||||
/// tell every client the current keepsake just became undownloadable.
|
||||
///
|
||||
/// The SSE matters: bumping the epoch retires the archive INSTANTLY, so `/export/zip` starts 404ing
|
||||
/// the moment the change commits. Without a nudge, a guest sitting on `/export` keeps rendering an
|
||||
/// enabled "download" button for the whole rebuild. Both the nav badge and the export page already
|
||||
/// refetch `/export/status` on `export-progress`, so a 0% tick is the cheapest correct signal.
|
||||
pub fn start_regen(state: &AppState, regen: crate::services::export::PendingRegen) {
|
||||
for export_type in ["zip", "html"] {
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"export-progress",
|
||||
serde_json::json!({ "type": export_type, "progress_pct": 0 }).to_string(),
|
||||
));
|
||||
}
|
||||
crate::services::export::spawn_export_jobs(
|
||||
regen.event_id,
|
||||
regen.event_name,
|
||||
regen.epoch,
|
||||
state.config.comments_enabled,
|
||||
// Debounced: a takedown pass is a burst, and each request retires the last generation. The
|
||||
// delay lets superseded workers fail their claim and do zero work instead of each building
|
||||
// a full archive. See export::REGEN_DEBOUNCE.
|
||||
crate::services::export::REGEN_DEBOUNCE,
|
||||
state.pool.clone(),
|
||||
state.config.media_path.clone(),
|
||||
state.config.export_path.clone(),
|
||||
state.sse_tx.clone(),
|
||||
);
|
||||
}
|
||||
|
||||
pub async fn host_delete_upload(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(upload_id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
let upload = Upload::find_by_id(&state.pool, upload_id)
|
||||
let upload = Upload::find_by_id_and_event(&state.pool, upload_id, auth.event_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Upload nicht gefunden.".into()))?;
|
||||
|
||||
Upload::soft_delete(&state.pool, upload_id).await?;
|
||||
// The delete and the keepsake invalidation are ONE transaction: if the delete committed and the
|
||||
// invalidation didn't, the taken-down photo would stay downloadable forever and nothing would
|
||||
// notice (the keepsake still looks complete, and the host can no longer find the upload to retry).
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let deleted = Upload::soft_delete_in_event(&mut tx, upload_id, auth.event_id).await?;
|
||||
if !deleted {
|
||||
return Err(AppError::NotFound("Upload nicht gefunden.".into()));
|
||||
}
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
Affects::Both,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "upload-deleted".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload.id }).to_string(),
|
||||
});
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"upload-deleted",
|
||||
serde_json::json!({ "upload_id": upload.id }).to_string(),
|
||||
));
|
||||
if let Some(r) = regen {
|
||||
start_regen(&state, r);
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
event_id = %auth.event_id,
|
||||
upload_id = %upload.id,
|
||||
"host: host_delete_upload"
|
||||
);
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
pub async fn host_delete_comment(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
RequireHost(auth): RequireHost,
|
||||
Path(comment_id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
Comment::find_by_id(&state.pool, comment_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Kommentar nicht gefunden.".into()))?;
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let deleted = Comment::soft_delete_in_event(&mut tx, comment_id, auth.event_id).await?;
|
||||
if !deleted {
|
||||
return Err(AppError::NotFound("Kommentar nicht gefunden.".into()));
|
||||
}
|
||||
// Only the HTML viewer embeds comments — the ZIP is media-only, so it is carried forward rather
|
||||
// than rebuilt. Otherwise moderating one comment would 404 the photo download for minutes to
|
||||
// change nothing in it.
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
Affects::ViewerOnly,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
|
||||
Comment::soft_delete(&state.pool, comment_id).await?;
|
||||
let _ = state.sse_tx.send(SseEvent::new(
|
||||
"comment-deleted",
|
||||
serde_json::json!({ "comment_id": comment_id }).to_string(),
|
||||
));
|
||||
if let Some(r) = regen {
|
||||
start_regen(&state, r);
|
||||
}
|
||||
tracing::info!(
|
||||
actor_user_id = %auth.user_id,
|
||||
event_id = %auth.event_id,
|
||||
comment_id = %comment_id,
|
||||
"host: host_delete_comment"
|
||||
);
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
@@ -193,17 +690,18 @@ pub async fn close_event(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
sqlx::query(
|
||||
let result = sqlx::query(
|
||||
"UPDATE event SET uploads_locked_at = NOW() WHERE slug = $1 AND uploads_locked_at IS NULL",
|
||||
)
|
||||
.bind(&state.config.event_slug)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "event-closed".to_string(),
|
||||
data: "{}".to_string(),
|
||||
});
|
||||
// Only broadcast when this call actually flipped the lock — closing an
|
||||
// already-closed event is a no-op and shouldn't spam listeners.
|
||||
if result.rows_affected() > 0 {
|
||||
let _ = state.sse_tx.send(SseEvent::new("event-closed", "{}"));
|
||||
}
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
@@ -212,17 +710,29 @@ pub async fn open_event(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
sqlx::query(
|
||||
"UPDATE event SET uploads_locked_at = NULL WHERE slug = $1",
|
||||
// Reopening invalidates any prior release: the keepsake was snapshotted at release time, so
|
||||
// allowing new uploads afterwards would silently diverge the live feed from the frozen export.
|
||||
//
|
||||
// ONE statement retires the entire export generation. Bumping `export_epoch` in the same write
|
||||
// that clears `export_released_at` instantly invalidates every in-flight worker, every `done`
|
||||
// row and all readiness — because readiness is DERIVED from this epoch (migration 014), not
|
||||
// stored. There is no export_job row to touch and nothing to keep in sync, so this needs no
|
||||
// transaction. Any worker still streaming holds the old epoch and is now inert: its
|
||||
// epoch-guarded finalize matches nothing, and it discards its own output.
|
||||
let result = sqlx::query(
|
||||
"UPDATE event
|
||||
SET uploads_locked_at = NULL,
|
||||
export_released_at = NULL,
|
||||
export_epoch = export_epoch + 1
|
||||
WHERE slug = $1 AND (uploads_locked_at IS NOT NULL OR export_released_at IS NOT NULL)",
|
||||
)
|
||||
.bind(&state.config.event_slug)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "event-opened".to_string(),
|
||||
data: "{}".to_string(),
|
||||
});
|
||||
if result.rows_affected() > 0 {
|
||||
let _ = state.sse_tx.send(SseEvent::new("event-opened", "{}"));
|
||||
}
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
@@ -231,39 +741,113 @@ pub async fn release_gallery(
|
||||
State(state): State<AppState>,
|
||||
RequireHost(_auth): RequireHost,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
let event = Event::find_by_slug(&state.pool, &state.config.event_slug)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Event nicht gefunden.".into()))?;
|
||||
// The release claim, the epoch bump, the upload lock and BOTH job rows are written in ONE
|
||||
// transaction. Two reasons, both of which were live bugs:
|
||||
//
|
||||
// 1. Cancellation. This handler used to commit `export_released_at` and only THEN await the
|
||||
// enqueue. Axum drops the handler future when the client disconnects (closed tab, proxy
|
||||
// timeout), which left the event released and uploads locked with ZERO export_job rows and
|
||||
// no workers: downloads 404 forever and the host cannot retry, because release_gallery
|
||||
// rejects an already-released event ("bereits freigegeben"). Only a restart escaped it.
|
||||
// Now nothing is committed until every row is written, and the workers are spawned AFTER
|
||||
// the commit (a detached `tokio::spawn` survives cancellation).
|
||||
// 2. Atomicity vs. a concurrent reopen. Bumping the epoch in the same statement that sets
|
||||
// `export_released_at` means no worker and no reader can ever observe "released again but
|
||||
// the generation hasn't moved on yet" — the window every previous fix kept leaving open.
|
||||
//
|
||||
// Release also locks uploads in the same statement (release ⇒ lock), so the export snapshot is
|
||||
// taken against a frozen upload set. `COALESCE` preserves an earlier explicit lock time.
|
||||
let mut tx = state.pool.begin().await?;
|
||||
|
||||
if event.export_released_at.is_some() {
|
||||
return Err(AppError::BadRequest("Galerie wurde bereits freigegeben.".into()));
|
||||
}
|
||||
let claimed: Option<(Uuid, String, i64)> = sqlx::query_as(
|
||||
"UPDATE event
|
||||
SET export_released_at = NOW(),
|
||||
uploads_locked_at = COALESCE(uploads_locked_at, NOW()),
|
||||
export_epoch = export_epoch + 1
|
||||
WHERE slug = $1 AND export_released_at IS NULL
|
||||
RETURNING id, name, export_epoch",
|
||||
)
|
||||
.bind(&state.config.event_slug)
|
||||
.fetch_optional(&mut *tx)
|
||||
.await?;
|
||||
|
||||
sqlx::query("UPDATE event SET export_released_at = NOW() WHERE slug = $1")
|
||||
.bind(&state.config.event_slug)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
let Some((event_id, event_name, epoch)) = claimed else {
|
||||
// Distinguish "no such event" from "already released" for a clean error.
|
||||
let exists = Event::find_by_slug(&state.pool, &state.config.event_slug)
|
||||
.await?
|
||||
.is_some();
|
||||
return Err(if exists {
|
||||
AppError::BadRequest("Galerie wurde bereits freigegeben.".into())
|
||||
} else {
|
||||
AppError::NotFound("Event nicht gefunden.".into())
|
||||
});
|
||||
};
|
||||
|
||||
// Enqueue export jobs
|
||||
for export_type in ["zip", "html"] {
|
||||
sqlx::query(
|
||||
"INSERT INTO export_job (event_id, type) VALUES ($1, $2::export_type)
|
||||
ON CONFLICT (event_id, type) DO NOTHING",
|
||||
)
|
||||
.bind(event.id)
|
||||
.bind(export_type)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
}
|
||||
// Arm both types at THIS epoch. No "skip the type that's already ready" check any more: the
|
||||
// epoch bump above retired every prior generation, so there is nothing to preserve and nothing
|
||||
// to be fooled by. (That skip was how a stale ready flag used to suppress regeneration.)
|
||||
crate::services::export::enqueue_jobs_at_epoch(&mut tx, event_id, epoch).await?;
|
||||
|
||||
// Spawn export workers
|
||||
tx.commit().await?;
|
||||
|
||||
// Release locks uploads too — tell any open composer to flip to the locked UI live rather than
|
||||
// discovering it via a rejected upload.
|
||||
let _ = state.sse_tx.send(SseEvent::new("event-closed", "{}"));
|
||||
|
||||
// Detached — survives this handler being cancelled.
|
||||
crate::services::export::spawn_export_jobs(
|
||||
event.id,
|
||||
event.name,
|
||||
event_id,
|
||||
event_name,
|
||||
epoch,
|
||||
state.config.comments_enabled,
|
||||
std::time::Duration::ZERO,
|
||||
state.pool.clone(),
|
||||
state.config.media_path.clone(),
|
||||
state.config.export_path.clone(),
|
||||
state.sse_tx.clone(),
|
||||
);
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{LOW_DISK_FLOOR_BYTES, disk_is_low};
|
||||
|
||||
const GB: u64 = 1_000_000_000;
|
||||
|
||||
#[test]
|
||||
fn a_healthy_disk_with_room_for_the_keepsake_is_not_low() {
|
||||
assert!(!disk_is_low(40 * GB, 25 * GB));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_absolute_floor_fires_even_when_the_gallery_is_tiny() {
|
||||
// All three volumes share one filesystem, so running out doesn't degrade one subsystem —
|
||||
// Postgres stops being able to write and the event goes down. A 1 GB gallery would clear
|
||||
// the keepsake test comfortably; the floor is what catches this.
|
||||
assert!(disk_is_low(5 * GB, GB));
|
||||
assert!(disk_is_low(LOW_DISK_FLOOR_BYTES - 1, 0));
|
||||
assert!(!disk_is_low(LOW_DISK_FLOOR_BYTES, 0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plenty_of_space_is_still_low_when_the_keepsake_would_not_fit() {
|
||||
// THE case the fixed threshold misses, and the one that matters: 30 GB free is nowhere near
|
||||
// any floor, but a 30 GB gallery needs room for TWO archives. The host can act on this
|
||||
// before releasing; after releasing, they cannot.
|
||||
assert!(disk_is_low(30 * GB, 66 * GB));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_keepsake_trigger_is_exact_at_the_boundary() {
|
||||
assert!(!disk_is_low(66 * GB, 66 * GB), "exactly enough is enough");
|
||||
assert!(disk_is_low(66 * GB - 1, 66 * GB));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_gallery_needs_nothing_and_only_the_floor_applies() {
|
||||
assert!(!disk_is_low(11 * GB, 0));
|
||||
assert!(disk_is_low(9 * GB, 0));
|
||||
}
|
||||
}
|
||||
|
||||
114
backend/src/handlers/me.rs
Normal file
114
backend/src/handlers/me.rs
Normal file
@@ -0,0 +1,114 @@
|
||||
//! Endpoints scoped to the *current user*. Kept separate from `auth::handlers` because
|
||||
//! these aren't about acquiring / refreshing a session — they're about reading my own
|
||||
//! state once I'm already signed in.
|
||||
//!
|
||||
//! Current routes:
|
||||
//! - `GET /api/v1/me/context` — bundled profile + feature flags + privacy note. The
|
||||
//! account page loads this once on mount instead of issuing several round trips.
|
||||
//! - `GET /api/v1/me/quota` — live per-user storage quota estimate.
|
||||
|
||||
use axum::Json;
|
||||
use axum::extract::State;
|
||||
use serde::Serialize;
|
||||
|
||||
use crate::auth::middleware::AuthUser;
|
||||
use crate::error::AppError;
|
||||
use crate::handlers::upload::compute_storage_quota;
|
||||
use crate::models::user::{User, UserRole};
|
||||
use crate::services::config;
|
||||
use crate::state::AppState;
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct QuotaDto {
|
||||
pub enabled: bool,
|
||||
pub used_bytes: i64,
|
||||
pub limit_bytes: Option<i64>,
|
||||
pub active_uploaders: i64,
|
||||
pub free_disk_bytes: i64,
|
||||
}
|
||||
|
||||
pub async fn get_quota(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<Json<QuotaDto>, AppError> {
|
||||
let user = User::find_by_id(&state.pool, auth.user_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
let estimate = compute_storage_quota(&state).await;
|
||||
|
||||
// Raw server telemetry (free disk, concurrent uploader count) is staff-only — it
|
||||
// must never reach a guest, even though the guest upload UI no longer renders it.
|
||||
// A guest still gets their own `used`/`limit` so enforcement stays transparent to
|
||||
// the code paths that consume it; only the server-wide fields are zeroed.
|
||||
let is_staff = matches!(auth.role, UserRole::Host | UserRole::Admin);
|
||||
|
||||
Ok(Json(QuotaDto {
|
||||
enabled: estimate.limit_bytes.is_some(),
|
||||
used_bytes: user.total_upload_bytes,
|
||||
limit_bytes: estimate.limit_bytes,
|
||||
active_uploaders: if is_staff {
|
||||
estimate.active_uploaders
|
||||
} else {
|
||||
0
|
||||
},
|
||||
free_disk_bytes: if is_staff {
|
||||
estimate.free_disk_bytes
|
||||
} else {
|
||||
0
|
||||
},
|
||||
}))
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct MeContextDto {
|
||||
pub user_id: uuid::Uuid,
|
||||
pub display_name: String,
|
||||
pub role: String,
|
||||
/// Plain-text Datenschutzhinweis set by the admin. Empty string when not configured.
|
||||
pub privacy_note: String,
|
||||
pub quota_enabled: bool,
|
||||
pub storage_quota_enabled: bool,
|
||||
/// Uploads are locked (event closed) — the composer should show a locked state live
|
||||
/// instead of letting a guest compose an upload only to eat a 403.
|
||||
pub uploads_locked: bool,
|
||||
/// The gallery has been released and the export snapshotted — uploads are permanently
|
||||
/// closed for this run (release ⇒ lock, and reopening regenerates).
|
||||
pub gallery_released: bool,
|
||||
}
|
||||
|
||||
pub async fn get_context(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<Json<MeContextDto>, AppError> {
|
||||
let user = User::find_by_id(&state.pool, auth.user_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
|
||||
let privacy_note = config::get_str(&state.config_cache, "privacy_note", "").await;
|
||||
let quota_enabled = config::get_bool(&state.config_cache, "quota_enabled", true).await;
|
||||
let storage_quota_enabled =
|
||||
config::get_bool(&state.config_cache, "storage_quota_enabled", true).await;
|
||||
|
||||
let event =
|
||||
crate::models::event::Event::find_by_slug(&state.pool, &state.config.event_slug).await?;
|
||||
let uploads_locked = event
|
||||
.as_ref()
|
||||
.map(|e| e.uploads_locked_at.is_some())
|
||||
.unwrap_or(false);
|
||||
let gallery_released = event
|
||||
.as_ref()
|
||||
.map(|e| e.export_released_at.is_some())
|
||||
.unwrap_or(false);
|
||||
|
||||
Ok(Json(MeContextDto {
|
||||
user_id: user.id,
|
||||
display_name: user.display_name,
|
||||
role: user.role.as_str().to_string(),
|
||||
privacy_note,
|
||||
quota_enabled,
|
||||
storage_quota_enabled,
|
||||
uploads_locked,
|
||||
gallery_released,
|
||||
}))
|
||||
}
|
||||
@@ -1,6 +1,10 @@
|
||||
pub mod admin;
|
||||
pub mod feed;
|
||||
pub mod health;
|
||||
pub mod host;
|
||||
pub mod me;
|
||||
pub mod public;
|
||||
pub mod social;
|
||||
pub mod sse;
|
||||
pub mod test_admin;
|
||||
pub mod upload;
|
||||
|
||||
44
backend/src/handlers/public.rs
Normal file
44
backend/src/handlers/public.rs
Normal file
@@ -0,0 +1,44 @@
|
||||
//! Unauthenticated, read-only endpoints safe to expose before a user has joined.
|
||||
|
||||
use axum::Json;
|
||||
use axum::extract::State;
|
||||
use serde::Serialize;
|
||||
|
||||
use crate::services::config;
|
||||
use crate::state::AppState;
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct PublicEventDto {
|
||||
pub name: String,
|
||||
pub slug: String,
|
||||
/// Whether the comment feature is on (env `COMMENTS_ENABLED`). The frontend hides
|
||||
/// the whole comment UI when false; exposed here so even the pre-auth shell knows.
|
||||
pub comments_enabled: bool,
|
||||
/// Active colour theme. `preset` is an id the frontend maps to a palette (or
|
||||
/// "custom"); `primary`/`accent` are the `#rrggbb` seeds the ramps derive from.
|
||||
/// Resolved as DB-config override → env default. Public so the theme applies on
|
||||
/// the very first (pre-auth) paint without a flash.
|
||||
pub theme_preset: String,
|
||||
pub theme_primary: String,
|
||||
pub theme_accent: String,
|
||||
}
|
||||
|
||||
/// Public event identity + presentation config, used by the pre-auth join/recover
|
||||
/// screens (which event am I joining, what does it look like). Only non-user-scoped
|
||||
/// fields are exposed, so this is safe without a token. Identity comes straight from
|
||||
/// instance config; the theme is resolved from the runtime `config` table (admin UI)
|
||||
/// falling back to the env-seeded default.
|
||||
pub async fn get_public_event(State(state): State<AppState>) -> Json<PublicEventDto> {
|
||||
let cache = &state.config_cache;
|
||||
Json(PublicEventDto {
|
||||
name: state.config.event_name.clone(),
|
||||
slug: state.config.event_slug.clone(),
|
||||
comments_enabled: state.config.comments_enabled,
|
||||
theme_preset: config::get_str(cache, "theme_preset", &state.config.default_theme_preset)
|
||||
.await,
|
||||
theme_primary: config::get_str(cache, "theme_primary", &state.config.default_theme_primary)
|
||||
.await,
|
||||
theme_accent: config::get_str(cache, "theme_accent", &state.config.default_theme_accent)
|
||||
.await,
|
||||
})
|
||||
}
|
||||
@@ -1,20 +1,65 @@
|
||||
use axum::extract::{Path, State};
|
||||
use axum::http::StatusCode;
|
||||
use axum::Json;
|
||||
use serde::Deserialize;
|
||||
use axum::extract::{Path, Query, State};
|
||||
use axum::http::StatusCode;
|
||||
use chrono::{DateTime, Utc};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::auth::middleware::AuthUser;
|
||||
use crate::error::AppError;
|
||||
use crate::models::comment::{Comment, CommentDto};
|
||||
use crate::models::hashtag::{self, Hashtag};
|
||||
use crate::models::upload::Upload;
|
||||
use crate::services::config;
|
||||
use crate::state::AppState;
|
||||
|
||||
/// Throttle a social write. Keyed PER USER, like the feed and upload limits and for the same
|
||||
/// reason: at a venue every guest sits behind one NAT, so an IP key hands the whole party a
|
||||
/// single bucket and the most active guest starves everyone else.
|
||||
///
|
||||
/// These were the only mutating endpoints in the app with no limit at all — the coverage was
|
||||
/// asymmetric, not deliberately open. The ceiling is set well above anything a real guest
|
||||
/// produces; this bounds a script, not an enthusiastic double-tapper.
|
||||
async fn check_social_rate(state: &AppState, user_id: Uuid) -> Result<(), AppError> {
|
||||
let rate_limits_on = config::get_bool(&state.config_cache, "rate_limits_enabled", true).await;
|
||||
let social_rate_on = config::get_bool(&state.config_cache, "social_rate_enabled", true).await;
|
||||
if !(rate_limits_on && social_rate_on) {
|
||||
return Ok(());
|
||||
}
|
||||
let rate_limit = config::get_usize(&state.config_cache, "social_rate_per_min", 120).await;
|
||||
// ONE bucket across likes, comments and comment deletions. Separate buckets would let a
|
||||
// caller triple the aggregate write rate just by alternating between them.
|
||||
state
|
||||
.rate_limiter
|
||||
.check_with_retry(
|
||||
format!("social:{user_id}"),
|
||||
rate_limit,
|
||||
std::time::Duration::from_secs(60),
|
||||
)
|
||||
.map_err(|retry_after_secs| {
|
||||
AppError::TooManyRequests(
|
||||
"Zu viele Aktionen. Bitte warte kurz und versuche es erneut.".into(),
|
||||
Some(retry_after_secs),
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct LikeResponse {
|
||||
/// The caller's like state *after* this toggle. The client sets `liked_by_me` from
|
||||
/// this rather than blind-inverting local state — otherwise a second device (same
|
||||
/// recovered user) drifts, since the `like-update` broadcast only carries `like_count`.
|
||||
pub liked: bool,
|
||||
/// Fresh like count, or `null` if the (best-effort) count query hiccuped. The client
|
||||
/// keeps its current count when this is null rather than adopting a wrong number.
|
||||
pub like_count: Option<i64>,
|
||||
}
|
||||
|
||||
pub async fn toggle_like(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
Path(upload_id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
) -> Result<Json<LikeResponse>, AppError> {
|
||||
// Check if user is banned
|
||||
let user = crate::models::user::User::find_by_id(&state.pool, auth.user_id)
|
||||
.await?
|
||||
@@ -22,8 +67,19 @@ pub async fn toggle_like(
|
||||
if user.is_banned {
|
||||
return Err(AppError::Forbidden("Du bist gesperrt.".into()));
|
||||
}
|
||||
check_social_rate(&state, auth.user_id).await?;
|
||||
|
||||
// Try to insert; if conflict, delete (toggle)
|
||||
// Event-scope: the upload must belong to the caller's event (404 otherwise),
|
||||
// matching the host handlers' find_by_id_and_event pattern.
|
||||
Upload::find_by_id_and_event(&state.pool, upload_id, auth.event_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Upload nicht gefunden.".into()))?;
|
||||
|
||||
// NOTE: liking is intentionally allowed while the event is locked. Locking
|
||||
// ("Event schließen") freezes *new uploads* only — likes, comments and
|
||||
// browsing stay open (USER_JOURNEYS §9.3, FEATURES capability matrix).
|
||||
|
||||
// Try to insert; if conflict, delete (toggle). `liked` = the caller's state afterwards.
|
||||
let result = sqlx::query(
|
||||
"INSERT INTO \"like\" (upload_id, user_id) VALUES ($1, $2)
|
||||
ON CONFLICT (upload_id, user_id) DO NOTHING",
|
||||
@@ -33,7 +89,8 @@ pub async fn toggle_like(
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
if result.rows_affected() == 0 {
|
||||
let liked = result.rows_affected() > 0;
|
||||
if !liked {
|
||||
// Already liked — remove
|
||||
sqlx::query("DELETE FROM \"like\" WHERE upload_id = $1 AND user_id = $2")
|
||||
.bind(upload_id)
|
||||
@@ -42,21 +99,55 @@ pub async fn toggle_like(
|
||||
.await?;
|
||||
}
|
||||
|
||||
// Broadcast SSE
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "like-update".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id }).to_string(),
|
||||
});
|
||||
// Fresh count so feed clients can patch the single card in place instead of
|
||||
// refetching page 1 (mirrors v_feed.like_count = COUNT(DISTINCT user_id)). The like
|
||||
// itself is already committed, so a failed count must not fail the request — but we
|
||||
// also must NOT broadcast/return a bogus 0 (that would push like_count: 0 to every
|
||||
// client until the next event). On error we skip the broadcast and return null.
|
||||
let like_count = sqlx::query_scalar::<_, i64>(
|
||||
"SELECT COUNT(DISTINCT user_id) FROM \"like\" WHERE upload_id = $1",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.fetch_one(&state.pool)
|
||||
.await
|
||||
.ok();
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
if let Some(count) = like_count {
|
||||
// Broadcast the new count so other clients patch their card. Only `like_count` is
|
||||
// shared — each client's own `liked_by_me` only changes via its own toggle (which
|
||||
// now reads it straight from this response).
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "like-update".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id, "like_count": count }).to_string(),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(Json(LikeResponse { liked, like_count }))
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Default)]
|
||||
pub struct ListCommentsQuery {
|
||||
/// RFC3339 timestamp — return only comments older than this. Pass the
|
||||
/// `created_at` of the oldest currently-loaded comment to fetch the next
|
||||
/// older page.
|
||||
pub before: Option<DateTime<Utc>>,
|
||||
}
|
||||
|
||||
const COMMENT_PAGE_SIZE: i64 = 50;
|
||||
|
||||
pub async fn list_comments(
|
||||
State(state): State<AppState>,
|
||||
_auth: AuthUser,
|
||||
auth: AuthUser,
|
||||
Path(upload_id): Path<Uuid>,
|
||||
Query(q): Query<ListCommentsQuery>,
|
||||
) -> Result<Json<Vec<CommentDto>>, AppError> {
|
||||
let comments = Comment::list_for_upload(&state.pool, upload_id).await?;
|
||||
// Event-scope: only list comments for an upload in the caller's event.
|
||||
Upload::find_by_id_and_event(&state.pool, upload_id, auth.event_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Upload nicht gefunden.".into()))?;
|
||||
|
||||
let comments =
|
||||
Comment::list_for_upload(&state.pool, upload_id, q.before, COMMENT_PAGE_SIZE).await?;
|
||||
Ok(Json(comments))
|
||||
}
|
||||
|
||||
@@ -71,40 +162,72 @@ pub async fn add_comment(
|
||||
Path(upload_id): Path<Uuid>,
|
||||
Json(body): Json<AddCommentRequest>,
|
||||
) -> Result<(StatusCode, Json<CommentDto>), AppError> {
|
||||
// Comments can be disabled instance-wide (env COMMENTS_ENABLED). The frontend hides
|
||||
// the UI, but gate the API too so a stale client or direct call can't slip one in.
|
||||
if !state.config.comments_enabled {
|
||||
return Err(AppError::Forbidden("Kommentare sind deaktiviert.".into()));
|
||||
}
|
||||
|
||||
let user = crate::models::user::User::find_by_id(&state.pool, auth.user_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Benutzer nicht gefunden.".into()))?;
|
||||
if user.is_banned {
|
||||
return Err(AppError::Forbidden("Du bist gesperrt.".into()));
|
||||
}
|
||||
check_social_rate(&state, auth.user_id).await?;
|
||||
|
||||
// Event-scope: only comment on an upload that belongs to the caller's event.
|
||||
Upload::find_by_id_and_event(&state.pool, upload_id, auth.event_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Upload nicht gefunden.".into()))?;
|
||||
|
||||
// NOTE: commenting is intentionally allowed while the event is locked. Locking
|
||||
// freezes *new uploads* only — likes, comments and browsing stay open
|
||||
// (USER_JOURNEYS §9.3, FEATURES capability matrix).
|
||||
|
||||
let text = body.body.trim();
|
||||
if text.is_empty() || text.len() > 500 {
|
||||
let text_chars = text.chars().count();
|
||||
if text_chars == 0 || text_chars > 500 {
|
||||
return Err(AppError::BadRequest(
|
||||
"Kommentar muss zwischen 1 und 500 Zeichen lang sein.".into(),
|
||||
));
|
||||
}
|
||||
|
||||
let comment = Comment::create(&state.pool, upload_id, auth.user_id, text).await?;
|
||||
|
||||
// Process hashtags in comment body
|
||||
// Insert the comment and link its hashtags atomically, so a crash mid-loop
|
||||
// can't leave a committed comment with only some of its tags indexed.
|
||||
let tags = hashtag::extract_hashtags(text);
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let comment = Comment::create(&mut *tx, upload_id, auth.user_id, text).await?;
|
||||
for tag in &tags {
|
||||
let h = Hashtag::upsert(&state.pool, auth.event_id, tag).await?;
|
||||
let h = Hashtag::upsert(&mut *tx, auth.event_id, tag).await?;
|
||||
sqlx::query(
|
||||
"INSERT INTO comment_hashtag (comment_id, hashtag_id) VALUES ($1, $2) ON CONFLICT DO NOTHING",
|
||||
)
|
||||
.bind(comment.id)
|
||||
.bind(h.id)
|
||||
.execute(&state.pool)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
}
|
||||
tx.commit().await?;
|
||||
|
||||
// Broadcast SSE
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "new-comment".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id }).to_string(),
|
||||
});
|
||||
// Fresh count so feed clients can patch the single card in place instead of
|
||||
// refetching page 1 (mirrors v_feed.comment_count = COUNT(DISTINCT c.id); COUNT(*)
|
||||
// over the same deleted_at filter is identical since comment.id is the PK). The
|
||||
// count + broadcast are a UI optimisation — the comment is already committed, so a
|
||||
// failure here must not fail the request. Swallow the error and skip the broadcast.
|
||||
if let Ok(comment_count) = sqlx::query_scalar::<_, i64>(
|
||||
"SELECT COUNT(*) FROM comment WHERE upload_id = $1 AND deleted_at IS NULL",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.fetch_one(&state.pool)
|
||||
.await
|
||||
{
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent {
|
||||
event_type: "new-comment".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id, "comment_count": comment_count })
|
||||
.to_string(),
|
||||
});
|
||||
}
|
||||
|
||||
let dto = CommentDto {
|
||||
id: comment.id,
|
||||
@@ -123,6 +246,11 @@ pub async fn delete_comment(
|
||||
auth: AuthUser,
|
||||
Path(comment_id): Path<Uuid>,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
// Banned users keep read access but cannot mutate (USER_JOURNEYS §10).
|
||||
if auth.is_banned {
|
||||
return Err(AppError::Forbidden("Du bist gesperrt.".into()));
|
||||
}
|
||||
check_social_rate(&state, auth.user_id).await?;
|
||||
let comment = Comment::find_by_id(&state.pool, comment_id)
|
||||
.await?
|
||||
.ok_or_else(|| AppError::NotFound("Kommentar nicht gefunden.".into()))?;
|
||||
@@ -131,6 +259,27 @@ pub async fn delete_comment(
|
||||
return Err(AppError::Forbidden("Nur eigene Kommentare löschen.".into()));
|
||||
}
|
||||
|
||||
Comment::soft_delete(&state.pool, comment_id).await?;
|
||||
// Event-scope: soft_delete_in_event only matches comments whose upload is in
|
||||
// the caller's event, so a cross-event comment_id resolves to a 404 here.
|
||||
let mut tx = state.pool.begin().await?;
|
||||
let deleted = Comment::soft_delete_in_event(&mut tx, comment_id, auth.event_id).await?;
|
||||
if !deleted {
|
||||
return Err(AppError::NotFound("Kommentar nicht gefunden.".into()));
|
||||
}
|
||||
// Comments live only in the HTML viewer, so the ZIP is carried forward, not rebuilt.
|
||||
let regen = crate::services::export::invalidate_and_arm(
|
||||
&mut tx,
|
||||
&state.config.event_slug,
|
||||
crate::services::export::Affects::ViewerOnly,
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
if let Some(r) = regen {
|
||||
crate::handlers::host::start_regen(&state, r);
|
||||
}
|
||||
let _ = state.sse_tx.send(crate::state::SseEvent::new(
|
||||
"comment-deleted",
|
||||
serde_json::json!({ "comment_id": comment_id, "upload_id": comment.upload_id }).to_string(),
|
||||
));
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
@@ -1,47 +1,134 @@
|
||||
use std::convert::Infallible;
|
||||
use std::time::Duration;
|
||||
|
||||
use axum::Json;
|
||||
use axum::extract::{Query, State};
|
||||
use axum::response::sse::{Event, KeepAlive, Sse};
|
||||
use futures::stream::Stream;
|
||||
use serde::Deserialize;
|
||||
use tokio_stream::wrappers::BroadcastStream;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use tokio_stream::StreamExt;
|
||||
use tokio_stream::wrappers::BroadcastStream;
|
||||
use tokio_stream::wrappers::errors::BroadcastStreamRecvError;
|
||||
|
||||
use crate::auth::jwt;
|
||||
use crate::auth::middleware::AuthUser;
|
||||
use crate::error::AppError;
|
||||
use crate::models::session::Session;
|
||||
use crate::state::AppState;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct SseQuery {
|
||||
pub token: String,
|
||||
pub ticket: String,
|
||||
}
|
||||
|
||||
/// SSE stream endpoint. Accepts JWT via query param since EventSource
|
||||
/// doesn't support custom headers.
|
||||
#[derive(Serialize)]
|
||||
pub struct StreamTicketResponse {
|
||||
pub ticket: String,
|
||||
/// Server clock at mint time — the client seeds its SSE reconnect cursor from this
|
||||
/// instead of `new Date()`, so a skewed browser clock can't drop uploads. See
|
||||
/// `DeltaResponse::server_time`.
|
||||
pub server_time: chrono::DateTime<chrono::Utc>,
|
||||
}
|
||||
|
||||
/// Mint a short-lived single-use SSE ticket. The browser's `EventSource` cannot
|
||||
/// send an `Authorization` header, so the alternative used to be passing the JWT
|
||||
/// as `?token=...` — which leaks the bearer token into access logs, referer
|
||||
/// headers, and browser history. The client now exchanges its Bearer token for
|
||||
/// an opaque ticket via this endpoint and passes that on the stream open.
|
||||
pub async fn issue_ticket(
|
||||
State(state): State<AppState>,
|
||||
auth: AuthUser,
|
||||
) -> Result<Json<StreamTicketResponse>, AppError> {
|
||||
// The endpoint had no rate limit at all. Authentication is not a bound here: one valid
|
||||
// session could loop it freely. 60/min is far above a real client (one ticket per SSE
|
||||
// (re)connect, and reconnects are backed off) while capping a loop.
|
||||
if let Err(retry_after_secs) = state.rate_limiter.check_with_retry(
|
||||
format!("sse_ticket:{}", auth.user_id),
|
||||
60,
|
||||
Duration::from_secs(60),
|
||||
) {
|
||||
return Err(AppError::TooManyRequests(
|
||||
"Zu viele Verbindungsversuche. Bitte warte kurz.".into(),
|
||||
Some(retry_after_secs),
|
||||
));
|
||||
}
|
||||
|
||||
let ticket = state.sse_tickets.issue(auth.token_hash).ok_or_else(|| {
|
||||
AppError::ServiceUnavailable(
|
||||
"Server ist gerade ausgelastet. Live-Updates folgen in Kürze.".into(),
|
||||
Some(30),
|
||||
)
|
||||
})?;
|
||||
let server_time = sqlx::query_scalar("SELECT NOW()")
|
||||
.fetch_one(&state.pool)
|
||||
.await?;
|
||||
Ok(Json(StreamTicketResponse {
|
||||
ticket,
|
||||
server_time,
|
||||
}))
|
||||
}
|
||||
|
||||
/// SSE stream endpoint. Authenticates via a single-use ticket (see
|
||||
/// [`issue_ticket`]) — never the raw JWT.
|
||||
pub async fn stream(
|
||||
State(state): State<AppState>,
|
||||
Query(q): Query<SseQuery>,
|
||||
) -> Result<Sse<impl Stream<Item = Result<Event, Infallible>>>, AppError> {
|
||||
// Verify token
|
||||
let _claims = jwt::verify_token(&q.token, &state.config.jwt_secret)
|
||||
.map_err(|_| AppError::Unauthorized("Token ungültig.".into()))?;
|
||||
let token_hash = state
|
||||
.sse_tickets
|
||||
.consume(&q.ticket)
|
||||
.ok_or_else(|| AppError::Unauthorized("Ticket ungültig oder abgelaufen.".into()))?;
|
||||
|
||||
let token_hash = jwt::hash_token(&q.token);
|
||||
// NOTE: this authenticates via ticket→session, not the `AuthUser` extractor. The
|
||||
// SSE stream is read-only, and under the read-only-ban model (USER_JOURNEYS §10)
|
||||
// banned users retain read access — so both minting a ticket and holding a stream
|
||||
// open are intentionally allowed for banned users; only writes are blocked.
|
||||
Session::find_by_token_hash(&state.pool, &token_hash)
|
||||
.await
|
||||
.map_err(|e| AppError::Internal(e.into()))?
|
||||
.ok_or_else(|| AppError::Unauthorized("Sitzung nicht gefunden.".into()))?;
|
||||
|
||||
let rx = state.sse_tx.subscribe();
|
||||
let stream = BroadcastStream::new(rx).filter_map(|msg| match msg {
|
||||
let events = BroadcastStream::new(rx).filter_map(|msg| match msg {
|
||||
Ok(sse_event) => Some(Ok(Event::default()
|
||||
.event(sse_event.event_type)
|
||||
.data(sse_event.data))),
|
||||
Err(_) => None,
|
||||
// A consumer that falls behind the broadcast buffer would otherwise silently
|
||||
// lose events, leaving its feed permanently stale. Instead of dropping the
|
||||
// gap, tell the client to resync — the frontend responds by running a full
|
||||
// feed-delta fetch (which reconciles both new uploads and deletions).
|
||||
Err(BroadcastStreamRecvError::Lagged(n)) => {
|
||||
tracing::warn!("SSE consumer lagged, dropped {n} event(s); emitting resync");
|
||||
Some(Ok(Event::default().event("resync").data(n.to_string())))
|
||||
}
|
||||
});
|
||||
|
||||
// The session is only checked once at open. Re-validate it periodically so a
|
||||
// logged-out, expired, OR banned session's stream is closed rather than kept alive
|
||||
// until the client happens to disconnect. Bounds a stale stream to ~60s. NOTE: a ban is
|
||||
// deliberately read-only and does NOT revoke sessions (see `ban_user`), so the
|
||||
// `is_banned` re-read below is LOAD-BEARING — it is the only thing that stops a banned
|
||||
// user's live push stream. Do not remove it. Only a *definitive* gone/expired/banned
|
||||
// state ends the stream; a transient DB error just retries next tick.
|
||||
let pool = state.pool.clone();
|
||||
let session_hash = token_hash.clone();
|
||||
let session_gone = async move {
|
||||
let mut ticker = tokio::time::interval(Duration::from_secs(60));
|
||||
ticker.tick().await; // consume the immediate first tick
|
||||
loop {
|
||||
ticker.tick().await;
|
||||
match Session::find_user_by_token_hash(&pool, &session_hash).await {
|
||||
Ok(Some(user)) if !user.is_banned => {} // still valid — keep streaming
|
||||
Ok(Some(_)) => break, // banned — cut the stream
|
||||
Ok(None) => break, // logged out or expired — stop
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "SSE session revalidation query failed; retrying");
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
let stream = futures::StreamExt::take_until(events, Box::pin(session_gone));
|
||||
|
||||
Ok(Sse::new(stream).keep_alive(
|
||||
KeepAlive::new()
|
||||
.interval(Duration::from_secs(30))
|
||||
|
||||
133
backend/src/handlers/test_admin.rs
Normal file
133
backend/src/handlers/test_admin.rs
Normal file
@@ -0,0 +1,133 @@
|
||||
//! Test-only admin routes. **Compiled in always, but only registered when
|
||||
//! `EVENTSNAP_TEST_MODE=1` is set in the environment.** The route returns a hard
|
||||
//! 404 in production builds because [`crate::main`] skips registering the handler.
|
||||
//!
|
||||
//! These exist to give the Playwright E2E suite a quick "reset everything"
|
||||
//! escape hatch without forcing tests to maintain raw SQL fixtures or spin up a
|
||||
//! fresh database container per test.
|
||||
|
||||
use axum::extract::State;
|
||||
use axum::http::StatusCode;
|
||||
|
||||
use crate::auth::middleware::RequireAdmin;
|
||||
use crate::error::AppError;
|
||||
use crate::state::AppState;
|
||||
|
||||
/// Truncates every event-scoped table, wipes media on disk, and reseeds the `config`
|
||||
/// table: numeric values from the migration defaults, but every feature toggle forced
|
||||
/// OFF (production seeds them ON — see the note at the reseed below). Requires an admin
|
||||
/// JWT — even with `EVENTSNAP_TEST_MODE=1` it cannot be hit anonymously.
|
||||
pub async fn truncate_all(
|
||||
State(state): State<AppState>,
|
||||
RequireAdmin(_auth): RequireAdmin,
|
||||
) -> Result<StatusCode, AppError> {
|
||||
// Truncate in dependency order doesn't matter with CASCADE, but listing the
|
||||
// tables explicitly makes the blast radius obvious in code review.
|
||||
sqlx::query(
|
||||
r#"TRUNCATE
|
||||
comment_hashtag,
|
||||
upload_hashtag,
|
||||
hashtag,
|
||||
"like",
|
||||
comment,
|
||||
export_job,
|
||||
upload,
|
||||
session,
|
||||
"user",
|
||||
event,
|
||||
config
|
||||
RESTART IDENTITY CASCADE"#,
|
||||
)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
// Reseed config. The NUMERIC values mirror migrations 005/015/016/017/019; the BOOLEAN
|
||||
// toggles deliberately do NOT — migration 009 seeds every one of them `true`
|
||||
// (production), and this forces them `false` so the suite isn't fighting rate limits
|
||||
// and quotas it isn't testing.
|
||||
//
|
||||
// Be aware of what that costs: this runs as an auto-fixture before EVERY test, so no
|
||||
// test starts from production's config unless it explicitly turns a toggle back on
|
||||
// (02-upload/rate-limit, 07-adversarial/ddos, 01-auth/rate-limit-nat, …). That blind
|
||||
// spot is exactly why an entire class of per-IP limiter bugs went unnoticed: the
|
||||
// limiters were simply off. When adding a limiter or quota, add a spec that enables it.
|
||||
//
|
||||
// Kept in sync by hand because pulling SQL out of the migration files at runtime is
|
||||
// fragile — if you add a config key in a migration, add it here too.
|
||||
sqlx::query(
|
||||
r#"INSERT INTO config (key, value) VALUES
|
||||
('max_image_size_mb', '20'),
|
||||
('max_video_size_mb', '500'),
|
||||
('upload_rate_per_hour', '100'),
|
||||
('feed_rate_per_min', '60'),
|
||||
('export_rate_per_day', '3'),
|
||||
('join_ip_rate_per_min', '60'),
|
||||
('recover_ip_rate_per_min', '30'),
|
||||
('social_rate_per_min', '120'),
|
||||
('quota_tolerance', '0.75'),
|
||||
('estimated_guest_count', '100'),
|
||||
('compression_concurrency', '2'),
|
||||
('rate_limits_enabled', 'false'),
|
||||
('upload_rate_enabled', 'false'),
|
||||
('feed_rate_enabled', 'false'),
|
||||
('export_rate_enabled', 'false'),
|
||||
('join_rate_enabled', 'false'),
|
||||
('social_rate_enabled', 'false'),
|
||||
('admin_login_rate_enabled', 'false'),
|
||||
('quota_enabled', 'false'),
|
||||
('storage_quota_enabled', 'false'),
|
||||
('upload_count_quota_enabled', 'false'),
|
||||
('privacy_note', '')
|
||||
ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value"#,
|
||||
)
|
||||
.execute(&state.pool)
|
||||
.await?;
|
||||
|
||||
// Wipe media directory. Best-effort: if it doesn't exist, that's fine.
|
||||
let _ = tokio::fs::remove_dir_all(&state.config.media_path).await;
|
||||
let _ = tokio::fs::create_dir_all(&state.config.media_path).await;
|
||||
|
||||
// Wipe the export directory too. Exports moved OUT of media_path (CR2 fix), so
|
||||
// the media wipe above no longer covers them — without this a real export in
|
||||
// one test would leave Gallery.zip on disk and contaminate the next.
|
||||
let _ = tokio::fs::remove_dir_all(&state.config.export_path).await;
|
||||
let _ = tokio::fs::create_dir_all(&state.config.export_path).await;
|
||||
|
||||
// The rate limiter holds an in-memory HashMap; clear it so a previous test's
|
||||
// counters don't leak into the next one.
|
||||
state.rate_limiter.clear();
|
||||
|
||||
// The reseed above wrote the `config` table directly (bypassing patch_config), so
|
||||
// the cache must be invalidated too — otherwise the first request after a truncate
|
||||
// could serve the previous test's toggles.
|
||||
state.config_cache.invalidate();
|
||||
|
||||
// The other two in-memory singletons that TRUNCATE used to leave standing.
|
||||
//
|
||||
// `disk_cache` holds a free-space reading for up to its TTL. TRUNCATE has just deleted every
|
||||
// uploaded file, which materially changes free space — so without this the next test can
|
||||
// compute a storage quota from the PREVIOUS test's disk. That was harmless only while quotas
|
||||
// were globally disabled in e2e (they no longer are: see specs/02-upload/quota.spec.ts, which
|
||||
// steers the per-user limit off `free_disk_bytes`), i.e. two holes were masking each other.
|
||||
state.disk_cache.invalidate();
|
||||
|
||||
// `sse_tickets` maps a ticket to a session token hash. TRUNCATE deletes the sessions, so every
|
||||
// surviving ticket is a dangling reference to a user that no longer exists.
|
||||
state.sse_tickets.clear();
|
||||
|
||||
// Invalidate any in-flight/queued compression task spawned by the previous test. Without this a
|
||||
// task still waiting on the concurrency semaphore wakes AFTER this wipe, fails to find its
|
||||
// (now-deleted) file, and broadcasts upload-error/upload-deleted into the NEXT test's SSE
|
||||
// stream. (Export workers are already inert across a truncate: they are epoch-guarded on the
|
||||
// event row, and truncate gives the event a fresh random UUID, so their writes match nothing.)
|
||||
state.compression.bump_generation();
|
||||
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
/// Returns whether the truncate endpoint is enabled. Used by the e2e harness
|
||||
/// during global-setup to fail loud if the test backend was started without
|
||||
/// `EVENTSNAP_TEST_MODE=1`.
|
||||
pub fn is_test_mode() -> bool {
|
||||
std::env::var("EVENTSNAP_TEST_MODE").as_deref() == Ok("1")
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,8 +1,7 @@
|
||||
use anyhow::Result;
|
||||
use axum::Router;
|
||||
use axum::extract::DefaultBodyLimit;
|
||||
use axum::routing::{delete, get, patch, post};
|
||||
use axum::Router;
|
||||
use tower_http::services::ServeDir;
|
||||
use tower_http::trace::TraceLayer;
|
||||
use tracing_subscriber::{layer::SubscriberExt, util::SubscriberInitExt};
|
||||
|
||||
@@ -18,63 +17,203 @@ mod state;
|
||||
use config::AppConfig;
|
||||
use state::AppState;
|
||||
|
||||
/// Hard HTTP body cap for the upload endpoint (576 MiB). Backstop against
|
||||
/// memory-exhaustion; precise per-class size limits are enforced in the handler.
|
||||
const MAX_UPLOAD_BYTES: usize = 576 * 1024 * 1024;
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> Result<()> {
|
||||
dotenvy::dotenv().ok();
|
||||
|
||||
tracing_subscriber::registry()
|
||||
.with(tracing_subscriber::EnvFilter::try_from_default_env().unwrap_or_else(|_| {
|
||||
"eventsnap_backend=debug,tower_http=debug".into()
|
||||
}))
|
||||
.with(
|
||||
// `info`, not `debug`. A stock deploy sets RUST_LOG nowhere (it is absent from
|
||||
// .env.example and was absent from docker-compose.yml), so this fallback IS the
|
||||
// production level — and at `debug` the TraceLayer below emits a line per request
|
||||
// AND per response, into a log file that had no rotation. `tower_http=warn`
|
||||
// rather than `info` states the intent: those spans are diagnostics, not an
|
||||
// access log, and a future `DefaultOnResponse::new().level(Level::INFO)` should
|
||||
// not silently re-enable them.
|
||||
tracing_subscriber::EnvFilter::try_from_default_env()
|
||||
.unwrap_or_else(|_| "eventsnap_backend=info,tower_http=warn".into()),
|
||||
)
|
||||
.with(tracing_subscriber::fmt::layer())
|
||||
.init();
|
||||
|
||||
let config = AppConfig::from_env()?;
|
||||
let pool = db::create_pool(&config.database_url).await?;
|
||||
let state = AppState::new(pool, config.clone());
|
||||
|
||||
// Reset any rows left mid-flight by a previous (possibly crashed) instance —
|
||||
// stuck `compression_status='processing'` uploads and `status='running'` export
|
||||
// jobs. Must run before the server starts taking requests so clients never see
|
||||
// the half-state.
|
||||
services::maintenance::startup_recovery(&pool).await;
|
||||
|
||||
let state = AppState::new(pool.clone(), config.clone());
|
||||
|
||||
// Regenerate image derivatives an older pipeline produced: the big-screen display for
|
||||
// uploads processed before it existed (v0.17.x), and anything predating the current
|
||||
// DERIVATIVES_REV (rev 1 applies the EXIF orientation, without which every portrait
|
||||
// phone photo is stored sideways). Fire-and-forget behind the compression semaphore;
|
||||
// originals are never touched, so a failure just retries on the next start.
|
||||
state.compression.backfill_stale_derivatives().await;
|
||||
|
||||
// Re-extract poster frames for videos a restart interrupted. `startup_recovery` above
|
||||
// marks their compression `failed` but nothing re-enqueued them, so `thumbnail_path`
|
||||
// stayed NULL for the rest of the event. Shares the attempt budget with the image
|
||||
// backfill, so a clip that genuinely yields no frame stops being retried.
|
||||
state.compression.backfill_video_posters().await;
|
||||
|
||||
// Re-spawn exports for events that were released but whose keepsake never finished
|
||||
// (crash mid-export). Needs the media/export paths + SSE sender, so it runs here
|
||||
// rather than inside `startup_recovery`. Fire-and-forget: the workers run in the
|
||||
// background; the HTTP server can start accepting requests meanwhile.
|
||||
services::export::recover_exports(
|
||||
pool.clone(),
|
||||
config.media_path.clone(),
|
||||
config.export_path.clone(),
|
||||
config.comments_enabled,
|
||||
state.sse_tx.clone(),
|
||||
)
|
||||
.await;
|
||||
|
||||
// Hourly background hygiene: prune expired sessions, evict cold rate-limiter
|
||||
// keys. Keeps the DB and process from growing unboundedly over multi-day events.
|
||||
services::maintenance::spawn_periodic_tasks(
|
||||
pool,
|
||||
state.rate_limiter.clone(),
|
||||
state.sse_tickets.clone(),
|
||||
config.media_path.clone(),
|
||||
);
|
||||
|
||||
// Ensure media directories exist
|
||||
tokio::fs::create_dir_all(&config.media_path).await.ok();
|
||||
|
||||
let api = Router::new()
|
||||
// Auth
|
||||
.route("/api/v1/event", get(handlers::public::get_public_event))
|
||||
.route("/api/v1/join", post(auth::handlers::join))
|
||||
.route("/api/v1/recover", post(auth::handlers::recover))
|
||||
// Forgotten-PIN escape hatch: ask a host to reset it (unauthenticated, throttled).
|
||||
.route(
|
||||
"/api/v1/recover/request",
|
||||
post(auth::handlers::request_pin_reset),
|
||||
)
|
||||
.route("/api/v1/admin/login", post(auth::handlers::admin_login))
|
||||
.route("/api/v1/session", delete(auth::handlers::logout))
|
||||
// Upload — body limit disabled; size validation is done inside the handler
|
||||
.route("/api/v1/upload", post(handlers::upload::upload)
|
||||
.route_layer(DefaultBodyLimit::disable()))
|
||||
// "Sign out everywhere" — revoke all of the caller's sessions.
|
||||
.route("/api/v1/sessions", delete(auth::handlers::logout_all))
|
||||
// Upload — HTTP-level body cap as an OOM backstop. The handler still enforces
|
||||
// the precise per-class limits from DB config (max_image/video_size_mb); this
|
||||
// layer just stops a multi-GB body from being buffered into memory before that
|
||||
// check runs. Sized generously above the default 500 MB video limit + multipart
|
||||
// overhead — if an admin raises max_video_size_mb above this, bump MAX_UPLOAD_BYTES.
|
||||
.route(
|
||||
"/api/v1/upload",
|
||||
post(handlers::upload::upload).route_layer(DefaultBodyLimit::max(MAX_UPLOAD_BYTES)),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/upload/{id}",
|
||||
patch(handlers::upload::edit_upload).delete(handlers::upload::delete_upload),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/upload/{id}/original",
|
||||
get(handlers::upload::get_original),
|
||||
)
|
||||
// Preview/thumbnail variants are gated the same way as originals (visibility
|
||||
// check + direct /media block below) so moderation actually revokes access to
|
||||
// the displayed images, not just the full-res download.
|
||||
.route(
|
||||
"/api/v1/upload/{id}/preview",
|
||||
get(handlers::upload::get_preview),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/upload/{id}/display",
|
||||
get(handlers::upload::get_display),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/upload/{id}/thumbnail",
|
||||
get(handlers::upload::get_thumbnail),
|
||||
)
|
||||
// Current-user endpoints (live quota estimate, profile + privacy note bundle)
|
||||
.route("/api/v1/me/context", get(handlers::me::get_context))
|
||||
.route("/api/v1/me/quota", get(handlers::me::get_quota))
|
||||
// Feed
|
||||
.route("/api/v1/feed", get(handlers::feed::feed))
|
||||
.route("/api/v1/feed/delta", get(handlers::feed::feed_delta))
|
||||
.route("/api/v1/hashtags", get(handlers::feed::hashtags))
|
||||
// Social
|
||||
.route("/api/v1/upload/{id}/like", post(handlers::social::toggle_like))
|
||||
.route(
|
||||
"/api/v1/upload/{id}/like",
|
||||
post(handlers::social::toggle_like),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/upload/{id}/comments",
|
||||
get(handlers::social::list_comments).post(handlers::social::add_comment),
|
||||
)
|
||||
.route("/api/v1/comment/{id}", delete(handlers::social::delete_comment))
|
||||
.route(
|
||||
"/api/v1/comment/{id}",
|
||||
delete(handlers::social::delete_comment),
|
||||
)
|
||||
// SSE
|
||||
.route("/api/v1/stream", get(handlers::sse::stream))
|
||||
.route("/api/v1/stream/ticket", post(handlers::sse::issue_ticket))
|
||||
// Host Dashboard
|
||||
.route("/api/v1/host/event", get(handlers::host::get_event_status))
|
||||
.route("/api/v1/host/event/close", post(handlers::host::close_event))
|
||||
.route(
|
||||
"/api/v1/host/event/close",
|
||||
post(handlers::host::close_event),
|
||||
)
|
||||
.route("/api/v1/host/event/open", post(handlers::host::open_event))
|
||||
.route("/api/v1/host/gallery/release", post(handlers::host::release_gallery))
|
||||
.route(
|
||||
"/api/v1/host/gallery/release",
|
||||
post(handlers::host::release_gallery),
|
||||
)
|
||||
// Escape hatch: force a keepsake rebuild. Without it a failed export is terminal at runtime
|
||||
// (release_gallery refuses an already-released event; recovery only runs at boot).
|
||||
.route(
|
||||
"/api/v1/host/export/rebuild",
|
||||
post(handlers::host::rebuild_export),
|
||||
)
|
||||
.route("/api/v1/host/users", get(handlers::host::list_users))
|
||||
.route("/api/v1/host/users/{id}/ban", post(handlers::host::ban_user))
|
||||
.route("/api/v1/host/users/{id}/unban", post(handlers::host::unban_user))
|
||||
.route("/api/v1/host/users/{id}/role", patch(handlers::host::set_role))
|
||||
.route("/api/v1/host/upload/{id}", delete(handlers::host::host_delete_upload))
|
||||
.route("/api/v1/host/comment/{id}", delete(handlers::host::host_delete_comment))
|
||||
.route(
|
||||
"/api/v1/host/users/{id}/ban",
|
||||
post(handlers::host::ban_user),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/users/{id}/unban",
|
||||
post(handlers::host::unban_user),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/users/{id}/role",
|
||||
patch(handlers::host::set_role),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/users/{id}/pin-reset",
|
||||
post(handlers::host::reset_user_pin),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/pin-reset-requests",
|
||||
get(handlers::host::list_pin_reset_requests),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/pin-reset-requests/{id}",
|
||||
delete(handlers::host::dismiss_pin_reset_request),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/upload/{id}",
|
||||
delete(handlers::host::host_delete_upload),
|
||||
)
|
||||
.route(
|
||||
"/api/v1/host/comment/{id}",
|
||||
delete(handlers::host::host_delete_comment),
|
||||
)
|
||||
// Export (all authenticated users)
|
||||
.route("/api/v1/export/status", get(handlers::admin::export_status))
|
||||
.route(
|
||||
"/api/v1/export/ticket",
|
||||
post(handlers::admin::export_ticket),
|
||||
)
|
||||
.route("/api/v1/export/zip", get(handlers::admin::download_zip))
|
||||
.route("/api/v1/export/html", get(handlers::admin::download_html))
|
||||
// Admin Dashboard
|
||||
@@ -83,21 +222,125 @@ async fn main() -> Result<()> {
|
||||
"/api/v1/admin/config",
|
||||
get(handlers::admin::get_config).patch(handlers::admin::patch_config),
|
||||
)
|
||||
.route("/api/v1/admin/export/jobs", get(handlers::admin::get_export_jobs));
|
||||
.route(
|
||||
"/api/v1/admin/export/jobs",
|
||||
get(handlers::admin::get_export_jobs),
|
||||
);
|
||||
|
||||
// Serve media files from disk
|
||||
let media_service = ServeDir::new(&config.media_path);
|
||||
// Test-only route: a hard reset for the Playwright E2E harness. The handler
|
||||
// is compiled in always, but the route is only attached when
|
||||
// `EVENTSNAP_TEST_MODE=1`. In production the call returns 404 — the route
|
||||
// simply isn't there.
|
||||
let api = if handlers::test_admin::is_test_mode() {
|
||||
tracing::warn!(
|
||||
"EVENTSNAP_TEST_MODE=1 — registering /api/v1/admin/__truncate. \
|
||||
DO NOT enable this in production."
|
||||
);
|
||||
api.route(
|
||||
"/api/v1/admin/__truncate",
|
||||
post(handlers::test_admin::truncate_all),
|
||||
)
|
||||
} else {
|
||||
api
|
||||
};
|
||||
|
||||
// NOTE: media is deliberately NOT served over HTTP.
|
||||
//
|
||||
// Files live under `media_path` so the compression worker and the export job can read
|
||||
// them off disk, but nothing may pull them straight from `/media/**` — that bypasses
|
||||
// the visibility checks (soft-delete + ban-hide) that make a host takedown stick.
|
||||
// Every legitimate fetch goes through `/api/v1/upload/{id}/{original,preview,display,
|
||||
// thumbnail}`, which filter via `find_visible_media`; those are the only media URLs the
|
||||
// backend ever emits (see `handlers::feed`).
|
||||
//
|
||||
// This used to be a `ServeDir` on `/media` with four `nest_service` blockers on the
|
||||
// subtrees above it. That was bypassable: axum routes on the RAW path while `ServeDir`
|
||||
// percent-decodes afterwards, so `/media/%70reviews/{id}.jpg` missed every blocker,
|
||||
// fell through to the `ServeDir`, and was decoded back to `previews/` on disk — serving
|
||||
// a taken-down photo to anyone, unauthenticated. Any single escaped byte worked, in all
|
||||
// four subtrees. Deleting the route removes the vector outright rather than racing the
|
||||
// decoder; `/media/**` now 404s regardless of encoding.
|
||||
let router = Router::new()
|
||||
// Liveness. Stays dependency-free — the compose healthcheck gates Caddy's startup
|
||||
// on it, so anything that can fail transiently must NOT be in here.
|
||||
.route("/health", get(|| async { "ok" }))
|
||||
// Readiness. Touches the pool; for an external monitor, not for the compose gate.
|
||||
.route("/health/ready", get(handlers::health::ready))
|
||||
.merge(api)
|
||||
.nest_service("/media", media_service)
|
||||
.layer(TraceLayer::new_for_http())
|
||||
.with_state(state);
|
||||
|
||||
let listener = tokio::net::TcpListener::bind(("0.0.0.0", config.app_port)).await?;
|
||||
tracing::info!("listening on {}", listener.local_addr()?);
|
||||
axum::serve(listener, router).await?;
|
||||
// `into_make_service_with_connect_info` is required by the pre-auth handlers, which
|
||||
// extract `ConnectInfo<SocketAddr>` to use the peer address as the rate-limit key when
|
||||
// X-Forwarded-For is absent. Without it those extractors fail at runtime.
|
||||
axum::serve(
|
||||
listener,
|
||||
router.into_make_service_with_connect_info::<std::net::SocketAddr>(),
|
||||
)
|
||||
.with_graceful_shutdown(shutdown_signal())
|
||||
.await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Hard cap on how long we wait for in-flight connections to drain after a shutdown
|
||||
/// signal. Uploads (streamed to disk) finish in well under this; the cap exists because
|
||||
/// long-lived SSE streams never end on their own and would otherwise keep the graceful
|
||||
/// drain — and thus the process — pending until the orchestrator force-kills it.
|
||||
const SHUTDOWN_GRACE: std::time::Duration = std::time::Duration::from_secs(10);
|
||||
|
||||
/// Resolves on SIGINT (Ctrl-C) or SIGTERM (container stop / deploy). Letting
|
||||
/// `axum::serve` drain in-flight requests on this signal means a redeploy no longer
|
||||
/// truncates uploads mid-flight; any background compression/export half-states that a
|
||||
/// hard kill would leave are already reconciled by `startup_recovery` on the next boot.
|
||||
///
|
||||
/// Once the signal fires we also arm a detached backstop that force-exits after
|
||||
/// [`SHUTDOWN_GRACE`]. Without it, open SSE streams (which have no natural end) would
|
||||
/// hold the graceful drain open indefinitely; the backstop bounds shutdown regardless
|
||||
/// of the orchestrator's own kill timeout. If the drain completes first, `main` returns
|
||||
/// and the process exits before the timer ever fires.
|
||||
async fn shutdown_signal() {
|
||||
let ctrl_c = async {
|
||||
let _ = tokio::signal::ctrl_c().await;
|
||||
};
|
||||
|
||||
#[cfg(unix)]
|
||||
let terminate = async {
|
||||
match tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate()) {
|
||||
Ok(mut sig) => {
|
||||
sig.recv().await;
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "failed to install SIGTERM handler");
|
||||
// Never resolve — fall back to ctrl_c only.
|
||||
std::future::pending::<()>().await;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
#[cfg(not(unix))]
|
||||
let terminate = std::future::pending::<()>();
|
||||
|
||||
tokio::select! {
|
||||
_ = ctrl_c => {}
|
||||
_ = terminate => {}
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
"shutdown signal received, draining in-flight requests (max {}s)",
|
||||
SHUTDOWN_GRACE.as_secs()
|
||||
);
|
||||
|
||||
// Backstop: if the graceful drain is still blocked after the grace window (almost
|
||||
// always because SSE streams are still open), exit anyway so deploys aren't stalled.
|
||||
tokio::spawn(async {
|
||||
tokio::time::sleep(SHUTDOWN_GRACE).await;
|
||||
tracing::warn!(
|
||||
"graceful drain exceeded {}s (likely open SSE streams); forcing exit",
|
||||
SHUTDOWN_GRACE.as_secs()
|
||||
);
|
||||
std::process::exit(0);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -3,6 +3,10 @@ use serde::Serialize;
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Row shape for `comment`: every field is populated by sqlx from `SELECT *` / `RETURNING *`.
|
||||
// `deleted_at` is not read in Rust today (the soft-delete filter lives in SQL), but it is part of
|
||||
// the row and stays here so the struct keeps mirroring the table.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct Comment {
|
||||
pub id: Uuid,
|
||||
@@ -24,49 +28,87 @@ pub struct CommentDto {
|
||||
}
|
||||
|
||||
impl Comment {
|
||||
pub async fn create(
|
||||
pool: &PgPool,
|
||||
/// Takes any executor so the caller can insert the comment and link its
|
||||
/// hashtags inside a single transaction.
|
||||
pub async fn create<'e, E>(
|
||||
executor: E,
|
||||
upload_id: Uuid,
|
||||
user_id: Uuid,
|
||||
body: &str,
|
||||
) -> Result<Self, sqlx::Error> {
|
||||
) -> Result<Self, sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
sqlx::query_as::<_, Self>(
|
||||
"INSERT INTO comment (upload_id, user_id, body) VALUES ($1, $2, $3) RETURNING *",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.bind(user_id)
|
||||
.bind(body)
|
||||
.fetch_one(pool)
|
||||
.fetch_one(executor)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn list_for_upload(pool: &PgPool, upload_id: Uuid) -> Result<Vec<CommentDto>, sqlx::Error> {
|
||||
/// Paginated comment listing — returns up to `limit` rows in chronological
|
||||
/// order (oldest first). If `before` is set, only comments older than that
|
||||
/// timestamp are returned, enabling backward cursor pagination ("load
|
||||
/// earlier"). Without the LIMIT a hot post with thousands of comments could
|
||||
/// OOM the server on a single GET.
|
||||
pub async fn list_for_upload(
|
||||
pool: &PgPool,
|
||||
upload_id: Uuid,
|
||||
before: Option<DateTime<Utc>>,
|
||||
limit: i64,
|
||||
) -> Result<Vec<CommentDto>, sqlx::Error> {
|
||||
// Two-step: pick the newest `limit` rows older than `before`, then flip
|
||||
// them back into ascending order so the caller can render top-to-bottom.
|
||||
sqlx::query_as::<_, CommentDto>(
|
||||
"SELECT c.id, c.upload_id, c.user_id, u.display_name AS uploader_name, c.body, c.created_at
|
||||
FROM comment c
|
||||
JOIN \"user\" u ON u.id = c.user_id
|
||||
WHERE c.upload_id = $1 AND c.deleted_at IS NULL
|
||||
ORDER BY c.created_at ASC",
|
||||
"SELECT * FROM (
|
||||
SELECT c.id, c.upload_id, c.user_id, u.display_name AS uploader_name,
|
||||
c.body, c.created_at
|
||||
FROM comment c
|
||||
JOIN \"user\" u ON u.id = c.user_id
|
||||
WHERE c.upload_id = $1 AND c.deleted_at IS NULL
|
||||
AND ($2::timestamptz IS NULL OR c.created_at < $2)
|
||||
ORDER BY c.created_at DESC
|
||||
LIMIT $3
|
||||
) page
|
||||
ORDER BY created_at ASC",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.bind(before)
|
||||
.bind(limit)
|
||||
.fetch_all(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn find_by_id(pool: &PgPool, id: Uuid) -> Result<Option<Self>, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"SELECT * FROM comment WHERE id = $1 AND deleted_at IS NULL",
|
||||
)
|
||||
.bind(id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
sqlx::query_as::<_, Self>("SELECT * FROM comment WHERE id = $1 AND deleted_at IS NULL")
|
||||
.bind(id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn soft_delete(pool: &PgPool, id: Uuid) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE comment SET deleted_at = NOW() WHERE id = $1")
|
||||
.bind(id)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
/// Event-scoped soft delete. Returns `false` if the comment doesn't exist or belongs to a
|
||||
/// different event.
|
||||
/// Executor-generic so the delete and the keepsake regeneration can share one transaction
|
||||
/// (see `Upload::soft_delete_in_event` for why that must be atomic).
|
||||
pub async fn soft_delete_in_event(
|
||||
conn: &mut sqlx::PgConnection,
|
||||
id: Uuid,
|
||||
event_id: Uuid,
|
||||
) -> Result<bool, sqlx::Error> {
|
||||
let result = sqlx::query(
|
||||
"UPDATE comment
|
||||
SET deleted_at = NOW()
|
||||
WHERE id = $1
|
||||
AND deleted_at IS NULL
|
||||
AND upload_id IN (SELECT id FROM upload WHERE event_id = $2)",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(event_id)
|
||||
.execute(conn)
|
||||
.await?;
|
||||
Ok(result.rows_affected() > 0)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,11 @@ use chrono::{DateTime, Utc};
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Row shape for `event`: every field is populated by sqlx from `SELECT *` / `RETURNING *`. Several
|
||||
// (`slug`, `cover_image_path`, `export_epoch`, `created_at`) are not read through this struct today
|
||||
// — callers that need them query the column directly — but they are part of the row and stay here so
|
||||
// the struct keeps mirroring the table.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct Event {
|
||||
pub id: Uuid,
|
||||
@@ -11,8 +16,11 @@ pub struct Event {
|
||||
pub is_active: bool,
|
||||
pub uploads_locked_at: Option<DateTime<Utc>>,
|
||||
pub export_released_at: Option<DateTime<Utc>>,
|
||||
pub export_zip_ready: bool,
|
||||
pub export_html_ready: bool,
|
||||
/// Monotonic generation counter for the keepsake. Bumped in the SAME UPDATE as any change to
|
||||
/// `export_released_at` (release and reopen are its only writers). An export is downloadable
|
||||
/// iff a `done` `export_job` row carries this exact epoch — readiness is derived from that,
|
||||
/// never stored, so it cannot drift and no worker can resurrect it. See migration 014.
|
||||
pub export_epoch: i64,
|
||||
pub created_at: DateTime<Utc>,
|
||||
}
|
||||
|
||||
@@ -25,13 +33,11 @@ impl Event {
|
||||
}
|
||||
|
||||
pub async fn create(pool: &PgPool, slug: &str, name: &str) -> Result<Self, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"INSERT INTO event (slug, name) VALUES ($1, $2) RETURNING *",
|
||||
)
|
||||
.bind(slug)
|
||||
.bind(name)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
sqlx::query_as::<_, Self>("INSERT INTO event (slug, name) VALUES ($1, $2) RETURNING *")
|
||||
.bind(slug)
|
||||
.bind(name)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn find_or_create(
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Row shape for `hashtag`, populated by sqlx from `RETURNING *` in `upsert`. Callers only use
|
||||
// `id` today; `event_id`/`tag` are the rest of the row and stay part of the struct.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct Hashtag {
|
||||
pub id: Uuid,
|
||||
@@ -10,7 +12,13 @@ pub struct Hashtag {
|
||||
|
||||
impl Hashtag {
|
||||
/// Upsert a hashtag (insert if not exists, return existing if it does).
|
||||
pub async fn upsert(pool: &PgPool, event_id: Uuid, tag: &str) -> Result<Self, sqlx::Error> {
|
||||
///
|
||||
/// Takes any executor so callers can run it inside a transaction (atomic
|
||||
/// upload/comment writes) or standalone against the pool.
|
||||
pub async fn upsert<'e, E>(executor: E, event_id: Uuid, tag: &str) -> Result<Self, sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
let normalized = tag.trim().trim_start_matches('#').to_lowercase();
|
||||
sqlx::query_as::<_, Self>(
|
||||
"INSERT INTO hashtag (event_id, tag) VALUES ($1, $2)
|
||||
@@ -19,59 +27,124 @@ impl Hashtag {
|
||||
)
|
||||
.bind(event_id)
|
||||
.bind(&normalized)
|
||||
.fetch_one(pool)
|
||||
.fetch_one(executor)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn link_to_upload(
|
||||
pool: &PgPool,
|
||||
pub async fn link_to_upload<'e, E>(
|
||||
executor: E,
|
||||
upload_id: Uuid,
|
||||
hashtag_id: Uuid,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
) -> Result<(), sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
sqlx::query(
|
||||
"INSERT INTO upload_hashtag (upload_id, hashtag_id) VALUES ($1, $2)
|
||||
ON CONFLICT DO NOTHING",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.bind(hashtag_id)
|
||||
.execute(pool)
|
||||
.execute(executor)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn unlink_all_from_upload(
|
||||
pool: &PgPool,
|
||||
pub async fn unlink_all_from_upload<'e, E>(
|
||||
executor: E,
|
||||
upload_id: Uuid,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
) -> Result<(), sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
sqlx::query("DELETE FROM upload_hashtag WHERE upload_id = $1")
|
||||
.bind(upload_id)
|
||||
.execute(pool)
|
||||
.execute(executor)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn tags_for_upload(
|
||||
pool: &PgPool,
|
||||
upload_id: Uuid,
|
||||
) -> Result<Vec<String>, sqlx::Error> {
|
||||
let rows: Vec<(String,)> = sqlx::query_as(
|
||||
"SELECT h.tag FROM hashtag h
|
||||
JOIN upload_hashtag uh ON uh.hashtag_id = h.id
|
||||
WHERE uh.upload_id = $1
|
||||
ORDER BY h.tag",
|
||||
)
|
||||
.bind(upload_id)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
Ok(rows.into_iter().map(|r| r.0).collect())
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract #hashtags from text (caption or body).
|
||||
/// Extract `#hashtags` from text (caption or body). Tags are restricted to
|
||||
/// ASCII letters, digits, and underscore — emoji / punctuation / accented
|
||||
/// characters are rejected. This is deliberately strict: hashtags are an index
|
||||
/// (used for filtering and SQL JOINs), and tags like `#🎉` or `#!?` accumulate
|
||||
/// noise without helping anyone find content.
|
||||
pub fn extract_hashtags(text: &str) -> Vec<String> {
|
||||
const MAX_TAG_LEN: usize = 40;
|
||||
text.split_whitespace()
|
||||
.filter(|w| w.starts_with('#') && w.len() > 1)
|
||||
.map(|w| w.trim_start_matches('#').to_lowercase())
|
||||
.filter(|t| !t.is_empty())
|
||||
.filter_map(|w| w.strip_prefix('#'))
|
||||
.map(|t| {
|
||||
t.chars()
|
||||
.take_while(|c| c.is_ascii_alphanumeric() || *c == '_')
|
||||
.collect::<String>()
|
||||
.to_lowercase()
|
||||
})
|
||||
.filter(|t| !t.is_empty() && t.chars().count() <= MAX_TAG_LEN)
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::extract_hashtags;
|
||||
|
||||
#[test]
|
||||
fn ascii_word_chars_extracted() {
|
||||
assert_eq!(
|
||||
extract_hashtags("hello #wedding #Day_2 #cake!"),
|
||||
vec!["wedding", "day_2", "cake"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn emojis_and_punctuation_excluded() {
|
||||
// The 🎉 tag drops out entirely (no ASCII chars after #), the next tag
|
||||
// stops at the !? and yields only "fun".
|
||||
assert_eq!(extract_hashtags("#🎉 #!? #fun!"), vec!["fun"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_or_bare_hash_skipped() {
|
||||
assert_eq!(extract_hashtags("# #"), Vec::<String>::new());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tag_stops_at_first_non_word_char() {
|
||||
// A tag runs until the first char that isn't ascii-alphanumeric or '_'.
|
||||
assert_eq!(extract_hashtags("#foo#bar"), vec!["foo"]);
|
||||
assert_eq!(extract_hashtags("#foo-bar"), vec!["foo"]);
|
||||
assert_eq!(extract_hashtags("##tag"), Vec::<String>::new()); // '#' after the strip is non-word
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tag_length_is_capped_at_40_chars() {
|
||||
let ok = "a".repeat(40);
|
||||
assert_eq!(extract_hashtags(&format!("#{ok}")), vec![ok.clone()]);
|
||||
// 41+ chars → dropped entirely (not truncated).
|
||||
let too_long = "a".repeat(41);
|
||||
assert_eq!(
|
||||
extract_hashtags(&format!("#{too_long}")),
|
||||
Vec::<String>::new()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn duplicate_tags_are_returned_verbatim_not_deduplicated() {
|
||||
// Dedup is the DB's job (Hashtag::upsert ON CONFLICT); extraction returns each
|
||||
// occurrence so callers can count/link them independently. Case folds to lower.
|
||||
assert_eq!(
|
||||
extract_hashtags("#fun #Fun #fun!"),
|
||||
vec!["fun", "fun", "fun"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_ascii_word_chars_truncate_the_tag() {
|
||||
// KNOWN LIMITATION for a German app: `is_ascii_alphanumeric` excludes umlauts
|
||||
// and ß, so a tag truncates at the first non-ASCII letter. Pinned here so a
|
||||
// future Unicode-aware change is a deliberate, test-visible decision.
|
||||
assert_eq!(extract_hashtags("#Grüße"), vec!["gr"]);
|
||||
assert_eq!(extract_hashtags("#Straße"), vec!["stra"]);
|
||||
assert_eq!(extract_hashtags("#café"), vec!["caf"]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,9 @@ use chrono::{DateTime, Utc};
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Row shape for `session`, populated by sqlx from `RETURNING *`. Session validation is done in SQL
|
||||
// (expiry/last-seen predicates), so no field is read in Rust — the struct is the row's shape.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct Session {
|
||||
pub id: Uuid,
|
||||
@@ -43,22 +46,65 @@ impl Session {
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn touch(pool: &PgPool, id: Uuid) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE session SET last_seen_at = NOW() WHERE id = $1")
|
||||
.bind(id)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
/// Resolve a session token straight to its live user row in one round-trip.
|
||||
///
|
||||
/// The auth extractor runs on every authenticated request and used to do two
|
||||
/// sequential queries (session lookup, then user lookup); this collapses them into
|
||||
/// a single `session JOIN "user"`. The `expires_at` guard mirrors
|
||||
/// [`Self::find_by_token_hash`], and the user row is read live (role/ban are never
|
||||
/// trusted from the JWT). Returns `None` when the session is missing/expired.
|
||||
pub async fn find_user_by_token_hash(
|
||||
pool: &PgPool,
|
||||
token_hash: &str,
|
||||
) -> Result<Option<crate::models::user::User>, sqlx::Error> {
|
||||
sqlx::query_as::<_, crate::models::user::User>(
|
||||
"SELECT u.*
|
||||
FROM session s
|
||||
JOIN \"user\" u ON u.id = s.user_id
|
||||
WHERE s.token_hash = $1 AND s.expires_at > NOW()",
|
||||
)
|
||||
.bind(token_hash)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Touch `last_seen_at` AND slide `expires_at` forward by `expiry_days` from now, so
|
||||
/// an actively-used session never hits the fixed 30-day cliff (it renews on every
|
||||
/// authenticated request). An idle session still expires `expiry_days` after its last
|
||||
/// activity. Keyed by token hash so the auth extractor needs no prior id lookup.
|
||||
pub async fn touch_and_renew(
|
||||
pool: &PgPool,
|
||||
token_hash: &str,
|
||||
expiry_days: i64,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
sqlx::query(
|
||||
"UPDATE session
|
||||
SET last_seen_at = NOW(),
|
||||
expires_at = NOW() + ($2 || ' days')::interval
|
||||
WHERE token_hash = $1",
|
||||
)
|
||||
.bind(token_hash)
|
||||
.bind(expiry_days.to_string())
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn delete_by_token_hash(
|
||||
pool: &PgPool,
|
||||
token_hash: &str,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
pub async fn delete_by_token_hash(pool: &PgPool, token_hash: &str) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("DELETE FROM session WHERE token_hash = $1")
|
||||
.bind(token_hash)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Revoke every session for a user. Backs "sign out everywhere" and the forced
|
||||
/// re-auth after a host PIN-reset or a ban. Returns the number of sessions cleared.
|
||||
pub async fn delete_all_for_user(pool: &PgPool, user_id: Uuid) -> Result<u64, sqlx::Error> {
|
||||
let r = sqlx::query("DELETE FROM session WHERE user_id = $1")
|
||||
.bind(user_id)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(r.rows_affected())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,10 @@ use serde::Serialize;
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Row shape for `upload`: every field is populated by sqlx from `RETURNING *` in `create`. Callers
|
||||
// mostly use `id` and hand the rest to the compression/feed queries, so most fields are never read
|
||||
// through this struct — they stay here so it keeps mirroring the table.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct Upload {
|
||||
pub id: Uuid,
|
||||
@@ -14,6 +18,7 @@ pub struct Upload {
|
||||
pub mime_type: String,
|
||||
pub original_size_bytes: i64,
|
||||
pub caption: Option<String>,
|
||||
pub compression_status: String,
|
||||
pub created_at: DateTime<Utc>,
|
||||
pub deleted_at: Option<DateTime<Utc>>,
|
||||
}
|
||||
@@ -34,16 +39,32 @@ pub struct UploadDto {
|
||||
pub created_at: DateTime<Utc>,
|
||||
}
|
||||
|
||||
/// Minimal projection of an upload's on-disk file paths, used by the visibility-gated
|
||||
/// media aliases so they don't hydrate the entire `Upload` row per request.
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct VisibleMedia {
|
||||
pub original_path: String,
|
||||
pub preview_path: Option<String>,
|
||||
pub thumbnail_path: Option<String>,
|
||||
pub display_path: Option<String>,
|
||||
pub mime_type: String,
|
||||
}
|
||||
|
||||
impl Upload {
|
||||
pub async fn create(
|
||||
pool: &PgPool,
|
||||
/// Takes any executor so the caller can run it inside a transaction (atomic
|
||||
/// quota + insert) or standalone against the pool.
|
||||
pub async fn create<'e, E>(
|
||||
executor: E,
|
||||
event_id: Uuid,
|
||||
user_id: Uuid,
|
||||
original_path: &str,
|
||||
mime_type: &str,
|
||||
original_size_bytes: i64,
|
||||
caption: Option<&str>,
|
||||
) -> Result<Self, sqlx::Error> {
|
||||
) -> Result<Self, sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
sqlx::query_as::<_, Self>(
|
||||
"INSERT INTO upload (event_id, user_id, original_path, mime_type, original_size_bytes, caption)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)
|
||||
@@ -55,19 +76,52 @@ impl Upload {
|
||||
.bind(mime_type)
|
||||
.bind(original_size_bytes)
|
||||
.bind(caption)
|
||||
.fetch_one(pool)
|
||||
.fetch_one(executor)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn find_by_id(pool: &PgPool, id: Uuid) -> Result<Option<Self>, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"SELECT * FROM upload WHERE id = $1 AND deleted_at IS NULL",
|
||||
/// Lean lookup for the public media aliases (`get_original`/`get_preview`/
|
||||
/// `get_thumbnail`): returns ONLY the file paths + mime for a visible upload —
|
||||
/// excluding soft-deleted rows, hidden owners (`uploads_hidden`), and banned owners
|
||||
/// (`is_banned`) — the same filter `v_feed` applies. So moderation that removes a post
|
||||
/// from the feed also stops its original/preview/thumbnail from being pulled by UUID.
|
||||
///
|
||||
/// Selects four columns instead of the whole `Upload` row: this runs once per image
|
||||
/// per cache-miss on the media hot path, so we avoid hydrating fields the response
|
||||
/// never uses.
|
||||
pub async fn find_visible_media(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
) -> Result<Option<VisibleMedia>, sqlx::Error> {
|
||||
sqlx::query_as::<_, VisibleMedia>(
|
||||
"SELECT up.original_path, up.preview_path, up.thumbnail_path, up.display_path, up.mime_type
|
||||
FROM upload up
|
||||
JOIN \"user\" u ON u.id = up.user_id
|
||||
WHERE up.id = $1 AND up.deleted_at IS NULL
|
||||
AND u.uploads_hidden = false AND u.is_banned = false",
|
||||
)
|
||||
.bind(id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Event-scoped lookup used by host endpoints so a host of event A cannot
|
||||
/// reach uploads belonging to event B.
|
||||
pub async fn find_by_id_and_event(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
event_id: Uuid,
|
||||
) -> Result<Option<Self>, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"SELECT * FROM upload
|
||||
WHERE id = $1 AND event_id = $2 AND deleted_at IS NULL",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(event_id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn set_preview_path(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
@@ -81,6 +135,82 @@ impl Upload {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn set_display_path(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
display_path: &str,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE upload SET display_path = $2 WHERE id = $1")
|
||||
.bind(id)
|
||||
.bind(display_path)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Stamp which revision of the derivative pipeline produced this row's preview/display,
|
||||
/// so the startup backfill can find rows generated by an older one exactly once.
|
||||
///
|
||||
/// Also clears the attempt counter: success is the only thing that resets it, and folding
|
||||
/// the reset in here means both the live path and the backfill get it with no extra call
|
||||
/// site to forget.
|
||||
pub async fn set_derivatives_rev(pool: &PgPool, id: Uuid, rev: i16) -> Result<(), sqlx::Error> {
|
||||
sqlx::query(
|
||||
"UPDATE upload
|
||||
SET derivatives_rev = $2, derivative_attempts = 0, derivative_last_error = NULL
|
||||
WHERE id = $1",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(rev)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Record that derivative processing is ABOUT to be attempted, returning the new count.
|
||||
///
|
||||
/// WRITE-AHEAD ON PURPOSE. The failure this bounds is a cgroup SIGKILL: the process
|
||||
/// vanishes mid-work, so no `Err` is returned, no error handler runs and no `Drop` fires.
|
||||
/// A counter incremented after a failure would increment zero times per crash and the
|
||||
/// boot loop would be unchanged. Counting the ATTEMPT is the only thing that survives the
|
||||
/// process dying. The cost is that a genuinely transient failure also burns an attempt —
|
||||
/// acceptable, because the retry budget is per-boot-loop, not per-request, and success
|
||||
/// resets it to zero.
|
||||
/// `None` when the row no longer exists (hard-deleted, or an e2e TRUNCATE landed while the
|
||||
/// task waited on the semaphore) — the caller should abandon quietly rather than treat a
|
||||
/// missing row as a processing failure.
|
||||
pub async fn begin_derivative_attempt(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
) -> Result<Option<i16>, sqlx::Error> {
|
||||
sqlx::query_scalar(
|
||||
"UPDATE upload
|
||||
SET derivative_attempts = derivative_attempts + 1
|
||||
WHERE id = $1
|
||||
RETURNING derivative_attempts",
|
||||
)
|
||||
.bind(id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Store why the last derivative attempt failed. Diagnostics only — nothing branches on it.
|
||||
pub async fn record_derivative_failure(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
error: &str,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
// Bounded: an anyhow chain can be long, and this is written on a failure path that may
|
||||
// repeat across every row of a bad batch.
|
||||
let truncated: String = error.chars().take(500).collect();
|
||||
sqlx::query("UPDATE upload SET derivative_last_error = $2 WHERE id = $1")
|
||||
.bind(id)
|
||||
.bind(truncated)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn set_thumbnail_path(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
@@ -94,22 +224,104 @@ impl Upload {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Soft-deletes the upload and decrements the uploader's `total_upload_bytes`.
|
||||
/// Done in a single transaction so a crash between the two writes can't leave
|
||||
/// the quota counter pointing at bytes the user has already deleted (which would
|
||||
/// silently lock them out of future uploads).
|
||||
///
|
||||
/// No-op if the row is already deleted — protects against a double-tap on the
|
||||
/// delete action double-decrementing the counter.
|
||||
pub async fn soft_delete(pool: &PgPool, id: Uuid) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE upload SET deleted_at = NOW() WHERE id = $1")
|
||||
let mut tx = pool.begin().await?;
|
||||
let row: Option<(Uuid, i64)> = sqlx::query_as(
|
||||
"UPDATE upload
|
||||
SET deleted_at = NOW()
|
||||
WHERE id = $1 AND deleted_at IS NULL
|
||||
RETURNING user_id, original_size_bytes",
|
||||
)
|
||||
.bind(id)
|
||||
.fetch_optional(&mut *tx)
|
||||
.await?;
|
||||
if let Some((user_id, bytes)) = row {
|
||||
sqlx::query(
|
||||
"UPDATE \"user\"
|
||||
SET total_upload_bytes = GREATEST(0, total_upload_bytes - $2)
|
||||
WHERE id = $1",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(bytes)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
}
|
||||
tx.commit().await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Event-scoped variant of [`Self::soft_delete`]. Returns `false` if no row
|
||||
/// matched (already deleted, wrong event, or unknown id) so host handlers
|
||||
/// can return a clean 404 instead of silently no-op'ing.
|
||||
/// Executor-generic so a caller can run the delete and the keepsake regeneration in ONE
|
||||
/// transaction. They must be atomic: if the delete commits and the regeneration doesn't (a
|
||||
/// dropped handler future, a failed second tx), the taken-down photo stays in the downloadable
|
||||
/// archive forever, and recovery can't tell — the keepsake still looks complete at the current
|
||||
/// epoch, and the host can no longer even find the upload to retry.
|
||||
pub async fn soft_delete_in_event(
|
||||
conn: &mut sqlx::PgConnection,
|
||||
id: Uuid,
|
||||
event_id: Uuid,
|
||||
) -> Result<bool, sqlx::Error> {
|
||||
let tx = conn;
|
||||
let row: Option<(Uuid, i64)> = sqlx::query_as(
|
||||
"UPDATE upload
|
||||
SET deleted_at = NOW()
|
||||
WHERE id = $1 AND event_id = $2 AND deleted_at IS NULL
|
||||
RETURNING user_id, original_size_bytes",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(event_id)
|
||||
.fetch_optional(&mut *tx)
|
||||
.await?;
|
||||
let deleted = if let Some((user_id, bytes)) = row {
|
||||
sqlx::query(
|
||||
"UPDATE \"user\"
|
||||
SET total_upload_bytes = GREATEST(0, total_upload_bytes - $2)
|
||||
WHERE id = $1",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(bytes)
|
||||
.execute(&mut *tx)
|
||||
.await?;
|
||||
true
|
||||
} else {
|
||||
false
|
||||
};
|
||||
Ok(deleted)
|
||||
}
|
||||
|
||||
pub async fn update_caption<'e, E>(
|
||||
executor: E,
|
||||
id: Uuid,
|
||||
caption: Option<&str>,
|
||||
) -> Result<(), sqlx::Error>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
sqlx::query("UPDATE upload SET caption = $2 WHERE id = $1")
|
||||
.bind(id)
|
||||
.execute(pool)
|
||||
.bind(caption)
|
||||
.execute(executor)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn update_caption(
|
||||
pub async fn set_compression_status(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
caption: Option<&str>,
|
||||
status: &str,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE upload SET caption = $2 WHERE id = $1")
|
||||
sqlx::query("UPDATE upload SET compression_status = $2 WHERE id = $1")
|
||||
.bind(id)
|
||||
.bind(caption)
|
||||
.bind(status)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
|
||||
@@ -12,6 +12,20 @@ pub enum UserRole {
|
||||
Admin,
|
||||
}
|
||||
|
||||
impl UserRole {
|
||||
pub fn as_str(&self) -> &'static str {
|
||||
match self {
|
||||
UserRole::Guest => "guest",
|
||||
UserRole::Host => "host",
|
||||
UserRole::Admin => "admin",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Row shape for `user`: every field is populated by sqlx from `SELECT *` / `RETURNING *`.
|
||||
// `uploads_hidden`, `failed_pin_attempts` and `created_at` are enforced/updated in SQL rather than
|
||||
// read in Rust, but they are part of the row and stay here so the struct keeps mirroring the table.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, sqlx::FromRow)]
|
||||
pub struct User {
|
||||
pub id: Uuid,
|
||||
@@ -46,6 +60,54 @@ impl User {
|
||||
.await
|
||||
}
|
||||
|
||||
/// Create a user with an explicit role, in ONE statement.
|
||||
///
|
||||
/// `create` + a separate `UPDATE ... SET role` is not equivalent: a crash or a pool error
|
||||
/// between the two leaves a GUEST row holding a reserved name, which is exactly the
|
||||
/// poisoned state that bricked admin login — now self-inflicted, and invisible to a
|
||||
/// role-based lookup, so the next login would create yet another.
|
||||
pub async fn create_with_role(
|
||||
pool: &PgPool,
|
||||
event_id: Uuid,
|
||||
display_name: &str,
|
||||
pin_hash: &str,
|
||||
role: UserRole,
|
||||
) -> Result<Self, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"INSERT INTO \"user\" (event_id, display_name, recovery_pin_hash, role)
|
||||
VALUES ($1, $2, $3, $4)
|
||||
RETURNING *",
|
||||
)
|
||||
.bind(event_id)
|
||||
.bind(display_name)
|
||||
.bind(pin_hash)
|
||||
.bind(role)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
/// The event's admin, looked up BY ROLE.
|
||||
///
|
||||
/// The name is not the identity and never was. Looking the admin up by `display_name`
|
||||
/// meant any guest who joined as "Admin" first made the lookup miss, and the fallback
|
||||
/// `create` then violated the case-insensitive unique index from migration 007 — a
|
||||
/// permanent 500 on admin login, recoverable only by hand-editing the database.
|
||||
///
|
||||
/// `ORDER BY created_at` so a database that somehow acquired two admin rows resolves to a
|
||||
/// stable one rather than alternating between them.
|
||||
pub async fn find_admin_for_event(
|
||||
pool: &PgPool,
|
||||
event_id: Uuid,
|
||||
) -> Result<Option<Self>, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>(
|
||||
"SELECT * FROM \"user\" WHERE event_id = $1 AND role = 'admin'
|
||||
ORDER BY created_at ASC LIMIT 1",
|
||||
)
|
||||
.bind(event_id)
|
||||
.fetch_optional(pool)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn find_by_id(pool: &PgPool, id: Uuid) -> Result<Option<Self>, sqlx::Error> {
|
||||
sqlx::query_as::<_, Self>("SELECT * FROM \"user\" WHERE id = $1")
|
||||
.bind(id)
|
||||
@@ -82,33 +144,54 @@ impl User {
|
||||
Ok(row.0)
|
||||
}
|
||||
|
||||
/// Window after which a failed-PIN streak is forgotten. Matches the lockout duration, so
|
||||
/// "wait out the cooldown" and "start clean" are the same interval to a guest.
|
||||
const PIN_ATTEMPT_DECAY_MINUTES: i64 = 15;
|
||||
|
||||
/// Record a wrong PIN and return the CURRENT streak length.
|
||||
///
|
||||
/// The counter decays: before this, it only ever cleared on a successful recovery or after
|
||||
/// a lockout expired, so ordinary typos accumulated across days and a guest could arrive at
|
||||
/// an event already most of the way to being locked out by mistakes made the night before.
|
||||
/// Decay is what makes the raised lock threshold safe rather than merely lenient.
|
||||
pub async fn increment_failed_pin(pool: &PgPool, id: Uuid) -> Result<i16, sqlx::Error> {
|
||||
let row: (i16,) = sqlx::query_as(
|
||||
"UPDATE \"user\"
|
||||
SET failed_pin_attempts = failed_pin_attempts + 1
|
||||
SET failed_pin_attempts = CASE
|
||||
WHEN last_failed_pin_at IS NULL
|
||||
OR last_failed_pin_at < NOW() - ($2 || ' minutes')::interval
|
||||
THEN 1
|
||||
ELSE failed_pin_attempts + 1
|
||||
END,
|
||||
last_failed_pin_at = NOW()
|
||||
WHERE id = $1
|
||||
RETURNING failed_pin_attempts",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(Self::PIN_ATTEMPT_DECAY_MINUTES.to_string())
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
Ok(row.0)
|
||||
}
|
||||
|
||||
pub async fn lock_pin(pool: &PgPool, id: Uuid, until: DateTime<Utc>) -> Result<(), sqlx::Error> {
|
||||
sqlx::query(
|
||||
"UPDATE \"user\" SET pin_locked_until = $2 WHERE id = $1",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(until)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
pub async fn lock_pin(
|
||||
pool: &PgPool,
|
||||
id: Uuid,
|
||||
until: DateTime<Utc>,
|
||||
) -> Result<(), sqlx::Error> {
|
||||
sqlx::query("UPDATE \"user\" SET pin_locked_until = $2 WHERE id = $1")
|
||||
.bind(id)
|
||||
.bind(until)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn reset_pin_attempts(pool: &PgPool, id: Uuid) -> Result<(), sqlx::Error> {
|
||||
sqlx::query(
|
||||
"UPDATE \"user\" SET failed_pin_attempts = 0, pin_locked_until = NULL WHERE id = $1",
|
||||
"UPDATE \"user\"
|
||||
SET failed_pin_attempts = 0, pin_locked_until = NULL, last_failed_pin_at = NULL
|
||||
WHERE id = $1",
|
||||
)
|
||||
.bind(id)
|
||||
.execute(pool)
|
||||
|
||||
@@ -1,36 +1,173 @@
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use sqlx::PgPool;
|
||||
use tokio::sync::Semaphore;
|
||||
use tokio::sync::{Semaphore, broadcast};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::models::upload::Upload;
|
||||
use crate::state::SseEvent;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct CompressionWorker {
|
||||
semaphore: Arc<Semaphore>,
|
||||
/// Serialises the memory-heavy image jobs — see `HEAVY_IMAGE_BYTES`. Separate from
|
||||
/// `semaphore` so ordinary photos keep full concurrency.
|
||||
heavy: Arc<Semaphore>,
|
||||
pool: PgPool,
|
||||
media_path: PathBuf,
|
||||
sse_tx: broadcast::Sender<SseEvent>,
|
||||
/// Bumped whenever the underlying data is reset out from under in-flight work (only the e2e
|
||||
/// TRUNCATE does this today). A task captures the value at spawn and abandons itself if it has
|
||||
/// changed by the time it runs — see `process`.
|
||||
generation: Arc<AtomicU64>,
|
||||
}
|
||||
|
||||
impl CompressionWorker {
|
||||
pub fn new(pool: PgPool, media_path: PathBuf, concurrency: usize) -> Self {
|
||||
pub fn new(
|
||||
pool: PgPool,
|
||||
media_path: PathBuf,
|
||||
concurrency: usize,
|
||||
sse_tx: broadcast::Sender<SseEvent>,
|
||||
) -> Self {
|
||||
Self {
|
||||
semaphore: Arc::new(Semaphore::new(concurrency)),
|
||||
heavy: Arc::new(Semaphore::new(1)),
|
||||
pool,
|
||||
media_path,
|
||||
sse_tx,
|
||||
generation: Arc::new(AtomicU64::new(0)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Invalidate all in-flight and queued compression work. Called by the e2e TRUNCATE endpoint:
|
||||
/// truncating deletes the upload rows and wipes `media/`, so a worker that was queued on the
|
||||
/// semaphore when the wipe happened would otherwise wake in the NEXT test, fail to find its
|
||||
/// file, and broadcast `upload-error` / `upload-deleted` into that test's live SSE stream —
|
||||
/// corrupting any test that asserts on toasts or feed contents. Bumping the generation makes
|
||||
/// those stale tasks return silently instead. A no-op in production (never called there).
|
||||
pub fn bump_generation(&self) {
|
||||
self.generation.fetch_add(1, Ordering::SeqCst);
|
||||
}
|
||||
|
||||
/// How many times `do_process` is attempted before an upload is given up on. The
|
||||
/// give-up path is user-visible (the photo disappears), so transient infrastructure
|
||||
/// errors must not reach it.
|
||||
const MAX_PROCESS_ATTEMPTS: u32 = 3;
|
||||
|
||||
/// Revision of the image-derivative pipeline. Bump this whenever a change makes existing
|
||||
/// previews/displays wrong, so `backfill_stale_derivatives` regenerates them once on the
|
||||
/// next start. Rev 1 = EXIF orientation is applied.
|
||||
const DERIVATIVES_REV: i16 = 1;
|
||||
|
||||
/// How many times derivative generation may be ATTEMPTED for one upload before it is left
|
||||
/// alone. Counted write-ahead and reset on success — see `Upload::begin_derivative_attempt`.
|
||||
///
|
||||
/// This is what turns a fatal input from an outage into a blemish. The startup backfill
|
||||
/// runs unconditionally on every boot, so before this bound a row whose processing killed
|
||||
/// the process was re-selected and re-run forever, and `restart: unless-stopped` made that
|
||||
/// an infinite loop that also dropped every SSE stream and truncated every in-flight
|
||||
/// upload on each cycle. Three attempts absorbs genuinely transient infrastructure
|
||||
/// failures (an ENOSPC spike, a pool blip) without ever becoming unbounded.
|
||||
const MAX_DERIVATIVE_ATTEMPTS: i16 = 3;
|
||||
|
||||
/// Rows regenerated per boot. Bounds both the query and the amount of work a single start
|
||||
/// can queue; whatever is left is picked up on the next boot.
|
||||
const BACKFILL_BATCH: i64 = 200;
|
||||
|
||||
/// Spawn a background task to process an uploaded file.
|
||||
pub fn process(&self, upload_id: Uuid, original_path: String, mime_type: String) {
|
||||
let worker = self.clone();
|
||||
let born_at = worker.generation.load(Ordering::SeqCst);
|
||||
tokio::spawn(async move {
|
||||
let _permit = worker.semaphore.acquire().await;
|
||||
if let Err(e) = worker.do_process(upload_id, &original_path, &mime_type).await {
|
||||
tracing::error!("compression failed for upload {upload_id}: {e:#}");
|
||||
// The data this task was queued against may have been reset while it waited for a permit
|
||||
// (e2e TRUNCATE). If so, its file and row are gone; doing anything — including
|
||||
// broadcasting a failure — would leak into an unrelated test. Abandon quietly.
|
||||
if worker.generation.load(Ordering::SeqCst) != born_at {
|
||||
return;
|
||||
}
|
||||
// Retry before giving up. Most failures here are transient and self-clearing —
|
||||
// an ENOSPC spike while several guests upload at once, a momentary DB-pool
|
||||
// exhaustion, a panic inside the image codec — and the give-up path is
|
||||
// user-visible data loss, so it is worth a few seconds to avoid entering it.
|
||||
//
|
||||
// But only for failures that CAN clear. An image that exceeds the decode budget,
|
||||
// is corrupt, or is in an unsupported format fails identically on every attempt,
|
||||
// so retrying it just burns 2s + 4s of backoff and writes three near-identical
|
||||
// warnings before reaching the same conclusion. Give up on those immediately.
|
||||
let mut attempt = 1u32;
|
||||
let outcome = loop {
|
||||
match worker
|
||||
.do_process(upload_id, &original_path, &mime_type)
|
||||
.await
|
||||
{
|
||||
Ok(v) => break Ok(v),
|
||||
Err(e)
|
||||
if attempt < Self::MAX_PROCESS_ATTEMPTS
|
||||
&& !crate::services::imaging::is_permanent_image_error(&e) =>
|
||||
{
|
||||
tracing::warn!(
|
||||
error = ?e, %upload_id, attempt,
|
||||
"compression attempt failed; retrying"
|
||||
);
|
||||
tokio::time::sleep(std::time::Duration::from_secs(2u64.pow(attempt))).await;
|
||||
attempt += 1;
|
||||
// The data may have been reset while we slept (e2e TRUNCATE).
|
||||
if worker.generation.load(Ordering::SeqCst) != born_at {
|
||||
return;
|
||||
}
|
||||
}
|
||||
Err(e) => break Err(e),
|
||||
}
|
||||
};
|
||||
|
||||
match outcome {
|
||||
Ok(_) => {
|
||||
tracing::info!("compression completed for upload {upload_id}");
|
||||
let _ = worker.sse_tx.send(SseEvent {
|
||||
event_type: "upload-processed".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id }).to_string(),
|
||||
});
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::error!(
|
||||
"compression failed for upload {upload_id} after {attempt} attempt(s): {e:#}"
|
||||
);
|
||||
// Refund + soft-delete (one tx, so v_feed excludes it) so a failed
|
||||
// transcode doesn't leave a permanently broken feed card or silently
|
||||
// charge the uploader's quota. Then tell the uploader (upload-error
|
||||
// toast) and evict the card everywhere (upload-deleted).
|
||||
//
|
||||
// The ORIGINAL IS DELIBERATELY KEPT. This path used to `remove_file` it
|
||||
// unconditionally, which meant any transient error — a disk-full blip
|
||||
// while saving a derivative, a pool hiccup, a panic in the image codec —
|
||||
// irreversibly destroyed the guest's only copy of a photo they can never
|
||||
// retake. The row is only soft-deleted, so keeping the bytes makes the
|
||||
// upload fully recoverable; the file is orphaned rather than lost, and
|
||||
// the path is logged so it can be found. `backfill_stale_derivatives`
|
||||
// already refuses to destroy data on error for exactly this reason.
|
||||
let _ = Upload::set_compression_status(&worker.pool, upload_id, "failed").await;
|
||||
if let Err(del) = Upload::soft_delete(&worker.pool, upload_id).await {
|
||||
tracing::warn!(error = ?del, %upload_id, "failed to soft-delete after compression failure");
|
||||
}
|
||||
tracing::warn!(
|
||||
%upload_id,
|
||||
path = %worker.media_path.join(&original_path).display(),
|
||||
"original retained for recovery after compression failure"
|
||||
);
|
||||
let _ = worker.sse_tx.send(SseEvent {
|
||||
event_type: "upload-error".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id, "error": e.to_string() })
|
||||
.to_string(),
|
||||
});
|
||||
let _ = worker.sse_tx.send(SseEvent {
|
||||
event_type: "upload-deleted".to_string(),
|
||||
data: serde_json::json!({ "upload_id": upload_id }).to_string(),
|
||||
});
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -41,99 +178,628 @@ impl CompressionWorker {
|
||||
original_path: &str,
|
||||
mime_type: &str,
|
||||
) -> Result<()> {
|
||||
Upload::set_compression_status(&self.pool, upload_id, "processing").await?;
|
||||
|
||||
let original = self.media_path.join(original_path);
|
||||
|
||||
if mime_type.starts_with("image/") {
|
||||
let preview_rel = self.generate_image_preview(upload_id, &original, mime_type).await?;
|
||||
// Count the attempt BEFORE doing the work — see `begin_derivative_attempt`. If this
|
||||
// input is the one that kills the container, this write is the only record that
|
||||
// survives, and it is what stops the boot backfill replaying it forever.
|
||||
match Upload::begin_derivative_attempt(&self.pool, upload_id).await? {
|
||||
Some(attempts) if attempts > Self::MAX_DERIVATIVE_ATTEMPTS => {
|
||||
anyhow::bail!(
|
||||
"derivative generation gave up after {} attempt(s)",
|
||||
attempts - 1
|
||||
);
|
||||
}
|
||||
Some(_) => {}
|
||||
// The row vanished while this task waited on the semaphore. Nothing to do, and
|
||||
// reporting a failure would broadcast into a stream that no longer has a card.
|
||||
None => return Ok(()),
|
||||
}
|
||||
let (preview_rel, display_rel) = self
|
||||
.generate_image_derivatives(upload_id, &original, mime_type)
|
||||
.await?;
|
||||
Upload::set_preview_path(&self.pool, upload_id, &preview_rel).await?;
|
||||
tracing::info!("preview generated for upload {upload_id}");
|
||||
Upload::set_display_path(&self.pool, upload_id, &display_rel).await?;
|
||||
Upload::set_derivatives_rev(&self.pool, upload_id, Self::DERIVATIVES_REV).await?;
|
||||
tracing::info!("preview + display generated for upload {upload_id}");
|
||||
} else if mime_type.starts_with("video/") {
|
||||
let thumb_rel = self.generate_video_thumbnail(upload_id, &original).await?;
|
||||
Upload::set_thumbnail_path(&self.pool, upload_id, &thumb_rel).await?;
|
||||
tracing::info!("thumbnail generated for upload {upload_id}");
|
||||
// A missing poster must NOT fail the upload. `set_thumbnail_path` is only reached when
|
||||
// a file really exists, so `thumbnail_path` stays NULL otherwise — which every consumer
|
||||
// already handles (FeedListCard, VirtualFeed, LightboxModal are all null-safe).
|
||||
//
|
||||
// The `?` here used to hide the defect; making the check strict without also making
|
||||
// this non-fatal would have been far worse than the bug. Every clip of a second or less
|
||||
// would fail compression, exhaust its retries and be soft-deleted — a cosmetic defect
|
||||
// turned into data loss, on exactly the mis-tap/Live-Photo clips guests produce most.
|
||||
match self.generate_video_thumbnail(upload_id, &original).await? {
|
||||
Some(thumb_rel) => {
|
||||
Upload::set_thumbnail_path(&self.pool, upload_id, &thumb_rel).await?;
|
||||
tracing::info!("thumbnail generated for upload {upload_id}");
|
||||
}
|
||||
None => {
|
||||
tracing::warn!(
|
||||
%upload_id,
|
||||
"no poster frame could be extracted; the video keeps its own tile"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Upload::set_compression_status(&self.pool, upload_id, "done").await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn generate_image_preview(
|
||||
/// Longest edge of the big-screen "display" derivative used by the diashow. Sized to be
|
||||
/// sharp on 1080p/4K while staying bounded (a ~2048px JPEG decodes to ~16 MB — trivial
|
||||
/// for any kiosk, unlike a raw multi-thousand-pixel original).
|
||||
const DISPLAY_MAX_EDGE: u32 = 2048;
|
||||
/// Longest edge of the phone-feed "preview" (data-saver default).
|
||||
const PREVIEW_MAX_EDGE: u32 = 800;
|
||||
|
||||
/// Above this pixel count the PNG original is stored as uploaded, unoptimised.
|
||||
///
|
||||
/// oxipng's peak memory scales with PIXELS, not file size: it decodes the PNG itself and
|
||||
/// then evaluates row filters, each trial holding a full-size buffer. That is why a 2.82
|
||||
/// MiB file could measure 1250 MiB of peak RSS inside a 1 GiB container — smooth,
|
||||
/// synthetic content compresses to almost nothing on disk while still being 8000x8000.
|
||||
/// 8 MP covers every real phone photo; beyond it we decline the (lossless, cosmetic)
|
||||
/// saving rather than risk the OOM kill.
|
||||
const OXIPNG_MAX_PIXELS: u64 = 8_000_000;
|
||||
|
||||
/// Estimated peak heap above which an image job takes the exclusive `heavy` permit.
|
||||
///
|
||||
/// `compression_concurrency` (default 2) bounds how many jobs run at once, but says
|
||||
/// nothing about how much memory each one costs, and the container gets 1 GiB total. A
|
||||
/// single 8000x8000 original measures ~516 MiB peak even with the decode correctly scoped
|
||||
/// — two of those overlapping is 1032 MiB and another OOM kill, from nothing more exotic
|
||||
/// than two guests uploading big photos at the same moment.
|
||||
///
|
||||
/// 150 MiB sits far above a normal phone photo (a 12 MP JPEG costs ~50 MiB all-in) so the
|
||||
/// common path never serialises, and far below the point where two jobs stop fitting.
|
||||
/// Throughput is unaffected for everything except the rare giant, which is exactly the
|
||||
/// case that must not run in parallel with another giant.
|
||||
const HEAVY_IMAGE_BYTES: u64 = 150 * 1024 * 1024;
|
||||
|
||||
/// Wall-clock ceiling for one oxipng run.
|
||||
///
|
||||
/// Bounds TIME, NOT MEMORY — oxipng checks the deadline between trials, so a single trial
|
||||
/// still allocates in full. The pixel gate above and the sequential build (see
|
||||
/// `default-features = false` in Cargo.toml) are what bound memory. Do not treat this
|
||||
/// constant as the OOM fix.
|
||||
const OXIPNG_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
|
||||
|
||||
/// Decode the image ONCE and emit both derivatives — the 800px `preview` (phone feed)
|
||||
/// and the 2048px `display` (diashow). Returns `(preview_rel, display_rel)`.
|
||||
async fn generate_image_derivatives(
|
||||
&self,
|
||||
upload_id: Uuid,
|
||||
original: &Path,
|
||||
mime_type: &str,
|
||||
) -> Result<String> {
|
||||
) -> Result<(String, String)> {
|
||||
let previews_dir = self.media_path.join("previews");
|
||||
let displays_dir = self.media_path.join("displays");
|
||||
tokio::fs::create_dir_all(&previews_dir).await?;
|
||||
tokio::fs::create_dir_all(&displays_dir).await?;
|
||||
|
||||
let preview_filename = format!("{upload_id}.jpg");
|
||||
let preview_path = previews_dir.join(&preview_filename);
|
||||
let filename = format!("{upload_id}.jpg");
|
||||
let preview_path = previews_dir.join(&filename);
|
||||
let display_path = displays_dir.join(&filename);
|
||||
let original = original.to_path_buf();
|
||||
let preview_path_clone = preview_path.clone();
|
||||
let mime_owned = mime_type.to_string();
|
||||
|
||||
// Run blocking image operations in a spawn_blocking task
|
||||
tokio::task::spawn_blocking(move || -> Result<()> {
|
||||
let img = image::open(&original)
|
||||
.context("failed to open image")?;
|
||||
|
||||
// Resize to max 800px wide, preserving aspect ratio
|
||||
let preview = img.resize(800, 800, image::imageops::FilterType::Lanczos3);
|
||||
preview.save_with_format(&preview_path_clone, image::ImageFormat::Jpeg)
|
||||
.context("failed to save preview")?;
|
||||
|
||||
// If the original is PNG, try lossless compression in-place
|
||||
if mime_owned == "image/png" {
|
||||
let opts = oxipng::Options::from_preset(2);
|
||||
let _ = oxipng::optimize(
|
||||
&oxipng::InFile::Path(original),
|
||||
&oxipng::OutFile::Path {
|
||||
path: None,
|
||||
preserve_attrs: true,
|
||||
},
|
||||
&opts,
|
||||
// Estimate the peak from the HEADER (no pixels decoded — the same kind of cheap probe
|
||||
// the upload handler already does via `exceeds_decode_budget`) and, if this job is a
|
||||
// giant, take the exclusive permit so it cannot overlap another giant. Held for the
|
||||
// whole blocking section, released on drop including on error.
|
||||
let estimate =
|
||||
crate::services::imaging::estimated_processing_peak_bytes(&original, Self::DISPLAY_MAX_EDGE);
|
||||
let _heavy_permit = match estimate {
|
||||
Some(bytes) if bytes > Self::HEAVY_IMAGE_BYTES => {
|
||||
tracing::debug!(
|
||||
%upload_id,
|
||||
estimated_mib = bytes / (1024 * 1024),
|
||||
"waiting for the heavy-image permit"
|
||||
);
|
||||
Some(self.heavy.acquire().await)
|
||||
}
|
||||
_ => None,
|
||||
};
|
||||
|
||||
Ok(())
|
||||
// Run blocking image operations in a spawn_blocking task
|
||||
tokio::task::spawn_blocking(move || {
|
||||
write_image_derivatives(upload_id, &original, &mime_owned, &preview_path, &display_path)
|
||||
})
|
||||
.await??;
|
||||
|
||||
Ok(format!("previews/{preview_filename}"))
|
||||
Ok((
|
||||
format!("previews/{filename}"),
|
||||
format!("displays/{filename}"),
|
||||
))
|
||||
}
|
||||
|
||||
/// Regenerate image derivatives that an older pipeline produced. Fire-and-forget from
|
||||
/// startup; picks up two cases, both of which leave the ORIGINAL untouched:
|
||||
///
|
||||
/// - uploads processed before the `display` derivative existed (preview but no
|
||||
/// `display_path`), and
|
||||
/// - uploads whose derivatives predate `DERIVATIVES_REV` — currently rev 1, which applies
|
||||
/// the EXIF orientation. Everything generated before it is stored sideways for any
|
||||
/// portrait phone photo.
|
||||
///
|
||||
/// Unlike the failure path in `process`, a backfill error is logged and skipped — it must
|
||||
/// NEVER destroy or soft-delete an upload that already has a working preview.
|
||||
///
|
||||
/// Bounded in three ways, all of them load-bearing on a box that restarts itself:
|
||||
/// `derivative_attempts` stops a fatal row being replayed on every boot, `BACKFILL_BATCH`
|
||||
/// stops one start queueing unbounded work, and the whole thing runs as ONE task walking
|
||||
/// the rows sequentially rather than N tasks racing for the same semaphore.
|
||||
pub async fn backfill_stale_derivatives(&self) {
|
||||
// `original_path IS NOT NULL` was dead — the column is NOT NULL. What actually needs
|
||||
// excluding is the blanked path `cleanup_deleted_media` leaves behind.
|
||||
let rows = sqlx::query_as::<_, (Uuid, String, String)>(
|
||||
"SELECT id, original_path, mime_type FROM upload
|
||||
WHERE deleted_at IS NULL AND mime_type LIKE 'image/%'
|
||||
AND original_path <> ''
|
||||
AND derivative_attempts < $2
|
||||
AND (
|
||||
(display_path IS NULL AND preview_path IS NOT NULL)
|
||||
OR derivatives_rev < $1
|
||||
)
|
||||
ORDER BY created_at DESC
|
||||
LIMIT $3",
|
||||
)
|
||||
.bind(Self::DERIVATIVES_REV)
|
||||
.bind(Self::MAX_DERIVATIVE_ATTEMPTS)
|
||||
.bind(Self::BACKFILL_BATCH)
|
||||
.fetch_all(&self.pool)
|
||||
.await;
|
||||
let rows = match rows {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "derivative backfill query failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
self.report_exhausted_derivatives().await;
|
||||
|
||||
if rows.is_empty() {
|
||||
return;
|
||||
}
|
||||
tracing::info!("regenerating derivatives for {} upload(s)", rows.len());
|
||||
|
||||
// ONE task for the whole batch. The previous shape spawned a task per row, so a large
|
||||
// backlog created thousands of live tasks that each held a pool handle and queued on
|
||||
// the same two semaphore permits, competing with live uploads for the entire boot.
|
||||
let worker = self.clone();
|
||||
tokio::spawn(async move {
|
||||
for (id, original_path, mime_type) in rows {
|
||||
let _permit = worker.semaphore.acquire().await;
|
||||
// Write-ahead, exactly as in the live path: if this row is the one that kills
|
||||
// the process, this increment is the only thing that outlives the SIGKILL.
|
||||
match Upload::begin_derivative_attempt(&worker.pool, id).await {
|
||||
Ok(Some(n)) if n > Self::MAX_DERIVATIVE_ATTEMPTS => continue,
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => continue,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, %id, "could not record a backfill attempt; skipping");
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let original = worker.media_path.join(&original_path);
|
||||
match worker
|
||||
.generate_image_derivatives(id, &original, &mime_type)
|
||||
.await
|
||||
{
|
||||
Ok((preview_rel, display_rel)) => {
|
||||
let _ = Upload::set_preview_path(&worker.pool, id, &preview_rel).await;
|
||||
let _ = Upload::set_display_path(&worker.pool, id, &display_rel).await;
|
||||
// Clears derivative_attempts too, so a row that failed transiently is
|
||||
// not one boot closer to being abandoned.
|
||||
let _ =
|
||||
Upload::set_derivatives_rev(&worker.pool, id, Self::DERIVATIVES_REV)
|
||||
.await;
|
||||
tracing::info!("derivatives regenerated for upload {id}");
|
||||
}
|
||||
Err(e) => {
|
||||
// Leave the existing derivatives and the original intact; this row is
|
||||
// retried on the next start until its attempt budget runs out. The rev
|
||||
// stays behind, which is the marker that it still needs doing.
|
||||
tracing::warn!(error = ?e, %id, "derivative backfill failed; leaving as-is");
|
||||
let _ =
|
||||
Upload::record_derivative_failure(&worker.pool, id, &format!("{e:#}"))
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Re-extract poster frames for videos that never got one.
|
||||
///
|
||||
/// A video interrupted by a restart is stranded: `startup_recovery` flips its
|
||||
/// `compression_status` from `processing` to `failed` and nothing re-enqueues it, so
|
||||
/// `thumbnail_path` stays NULL forever while the clip itself plays fine. The feed shows a
|
||||
/// posterless tile for the rest of the event, and after
|
||||
/// `FAILED_ORIGINAL_RETENTION_DAYS` the reclaim sweep is entitled to the original.
|
||||
///
|
||||
/// Shares `derivative_attempts` with the image backfill on purpose. Note the consequence,
|
||||
/// which is intended rather than a bug to fix later: `extract_poster_frame` returning
|
||||
/// `Ok(false)` is a NORMAL, permanent outcome for a sub-second clip (Live Photos,
|
||||
/// mis-taps), and since the counter is write-ahead and only cleared by a real success,
|
||||
/// those clips stop being re-ffmpeg'd on every boot once the budget is spent.
|
||||
pub async fn backfill_video_posters(&self) {
|
||||
let rows = sqlx::query_as::<_, (Uuid, String)>(
|
||||
"SELECT id, original_path FROM upload
|
||||
WHERE deleted_at IS NULL AND mime_type LIKE 'video/%'
|
||||
AND thumbnail_path IS NULL
|
||||
AND original_path <> ''
|
||||
AND derivative_attempts < $1
|
||||
ORDER BY created_at DESC
|
||||
LIMIT $2",
|
||||
)
|
||||
.bind(Self::MAX_DERIVATIVE_ATTEMPTS)
|
||||
.bind(Self::BACKFILL_BATCH)
|
||||
.fetch_all(&self.pool)
|
||||
.await;
|
||||
let rows = match rows {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "video poster backfill query failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
if rows.is_empty() {
|
||||
return;
|
||||
}
|
||||
tracing::info!("re-extracting posters for {} video(s)", rows.len());
|
||||
|
||||
let worker = self.clone();
|
||||
tokio::spawn(async move {
|
||||
for (id, original_path) in rows {
|
||||
let _permit = worker.semaphore.acquire().await;
|
||||
match Upload::begin_derivative_attempt(&worker.pool, id).await {
|
||||
Ok(Some(n)) if n > Self::MAX_DERIVATIVE_ATTEMPTS => continue,
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => continue,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, %id, "could not record a poster attempt; skipping");
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let original = worker.media_path.join(&original_path);
|
||||
match worker.generate_video_thumbnail(id, &original).await {
|
||||
Ok(Some(thumb_rel)) => {
|
||||
if Upload::set_thumbnail_path(&worker.pool, id, &thumb_rel)
|
||||
.await
|
||||
.is_ok()
|
||||
{
|
||||
// Clears the attempt counter: a video that eventually succeeded
|
||||
// must not carry a budget scar into a future pipeline revision.
|
||||
let _ = Upload::set_derivatives_rev(
|
||||
&worker.pool,
|
||||
id,
|
||||
Self::DERIVATIVES_REV,
|
||||
)
|
||||
.await;
|
||||
tracing::info!("poster regenerated for upload {id}");
|
||||
}
|
||||
}
|
||||
// No frame at all — normal for a very short clip. The tile stays
|
||||
// posterless and the attempt is spent, which is what stops the retry.
|
||||
Ok(None) => {
|
||||
tracing::debug!(%id, "still no poster frame; leaving the tile as-is");
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, %id, "poster backfill failed; leaving as-is");
|
||||
let _ =
|
||||
Upload::record_derivative_failure(&worker.pool, id, &format!("{e:#}"))
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Say out loud, once per boot, that some uploads have stopped being retried.
|
||||
///
|
||||
/// Without this the give-up is invisible: the loop stops (which is the point) but the
|
||||
/// affected photos keep a stale or missing derivative forever with nothing to notice. The
|
||||
/// originals are untouched, so this is recoverable once the cause is fixed — reset
|
||||
/// `derivative_attempts` to 0 and restart.
|
||||
async fn report_exhausted_derivatives(&self) {
|
||||
let exhausted: Result<i64, _> = sqlx::query_scalar(
|
||||
"SELECT count(*) FROM upload
|
||||
WHERE deleted_at IS NULL AND mime_type LIKE 'image/%'
|
||||
AND derivative_attempts >= $2
|
||||
AND (
|
||||
(display_path IS NULL AND preview_path IS NOT NULL)
|
||||
OR derivatives_rev < $1
|
||||
)",
|
||||
)
|
||||
.bind(Self::DERIVATIVES_REV)
|
||||
.bind(Self::MAX_DERIVATIVE_ATTEMPTS)
|
||||
.fetch_one(&self.pool)
|
||||
.await;
|
||||
if let Ok(count) = exhausted
|
||||
&& count > 0
|
||||
{
|
||||
tracing::error!(
|
||||
count,
|
||||
"{count} upload(s) exhausted derivative regeneration and will no longer be \
|
||||
retried; their originals are intact — see upload.derivative_last_error, fix \
|
||||
the cause, then reset derivative_attempts to 0 and restart"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the feed poster for a video. `Ok(None)` when the clip yields no frame — see
|
||||
/// [`crate::services::video::extract_poster_frame`], which owns the seek order, the timeout and
|
||||
/// the artifact check that this function used to be missing.
|
||||
async fn generate_video_thumbnail(
|
||||
&self,
|
||||
upload_id: Uuid,
|
||||
original: &Path,
|
||||
) -> Result<String> {
|
||||
) -> Result<Option<String>> {
|
||||
let thumbs_dir = self.media_path.join("thumbnails");
|
||||
tokio::fs::create_dir_all(&thumbs_dir).await?;
|
||||
|
||||
let thumb_filename = format!("{upload_id}.jpg");
|
||||
let thumb_path = thumbs_dir.join(&thumb_filename);
|
||||
|
||||
let output = tokio::process::Command::new("ffmpeg")
|
||||
.args([
|
||||
"-i",
|
||||
original.to_str().unwrap_or_default(),
|
||||
"-vframes",
|
||||
"1",
|
||||
"-ss",
|
||||
"00:00:01",
|
||||
"-vf",
|
||||
"scale=800:-1",
|
||||
"-y",
|
||||
thumb_path.to_str().unwrap_or_default(),
|
||||
])
|
||||
.output()
|
||||
.await
|
||||
.context("failed to run ffmpeg")?;
|
||||
let produced =
|
||||
crate::services::video::extract_poster_frame(original, &thumb_path, 800).await?;
|
||||
|
||||
if !output.status.success() {
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
anyhow::bail!("ffmpeg failed: {stderr}");
|
||||
}
|
||||
|
||||
Ok(format!("thumbnails/{thumb_filename}"))
|
||||
Ok(produced.then(|| format!("thumbnails/{thumb_filename}")))
|
||||
}
|
||||
}
|
||||
|
||||
/// The blocking half of [`CompressionWorker::generate_image_derivatives`]: decode once, write
|
||||
/// both derivatives, then optionally shrink a PNG original in place.
|
||||
///
|
||||
/// A free function rather than an inline closure so its memory behaviour is directly testable —
|
||||
/// this is the code path that OOM-killed the container, and the fix is a scoping property that a
|
||||
/// future edit could silently undo.
|
||||
fn write_image_derivatives(
|
||||
upload_id: Uuid,
|
||||
original: &Path,
|
||||
mime_type: &str,
|
||||
preview_path: &Path,
|
||||
display_path: &Path,
|
||||
) -> Result<()> {
|
||||
let preview_max = CompressionWorker::PREVIEW_MAX_EDGE;
|
||||
let display_max = CompressionWorker::DISPLAY_MAX_EDGE;
|
||||
|
||||
// THE FULL-SIZE DECODE IS SCOPED TO THIS BLOCK ON PURPOSE, and the block yields the
|
||||
// DISPLAY derivative rather than the original.
|
||||
//
|
||||
// `img` is up to 256 MiB (imaging::decode_limits max_alloc) and `resize` only BORROWS it,
|
||||
// so it used to stay alive through both resizes AND the oxipng call below — which decodes
|
||||
// the PNG a second time and holds a full-size buffer per filter trial. That measured
|
||||
// ~1250 MiB of peak RSS for a 2.8 MiB input, inside a 1 GiB cgroup: the container was
|
||||
// SIGKILLed, taking every SSE stream and every in-flight upload with it.
|
||||
//
|
||||
// A block rather than a bare `drop(img)` because a `drop` call is one careless edit away
|
||||
// from being removed as redundant-looking — and note the `else` arm MOVES `img` out, which
|
||||
// is what makes "the block's value is the only survivor" true in both arms.
|
||||
let (display, width, height) = {
|
||||
// Decompression-bomb limits + EXIF orientation, both in one place — see
|
||||
// services::imaging for why neither may be skipped.
|
||||
let img = crate::services::imaging::decode_oriented(original)?;
|
||||
let (width, height) = (img.width(), img.height());
|
||||
|
||||
// Display: max 2048px for the diashow. Only DOWNSCALE — never upscale a smaller
|
||||
// original (that adds bytes with no quality gain); re-encode it as JPEG as-is.
|
||||
let display = if width > display_max || height > display_max {
|
||||
img.resize(
|
||||
display_max,
|
||||
display_max,
|
||||
image::imageops::FilterType::Lanczos3,
|
||||
)
|
||||
} else {
|
||||
img
|
||||
};
|
||||
(display, width, height)
|
||||
};
|
||||
|
||||
display
|
||||
.save_with_format(display_path, image::ImageFormat::Jpeg)
|
||||
.context("failed to save display")?;
|
||||
|
||||
// Preview: max 800px, derived from the DISPLAY, not from the original.
|
||||
//
|
||||
// Both derivatives used to resize the full-size decode independently, so a 8000x8000
|
||||
// original paid for two full-size Lanczos passes and their intermediates — measured 520
|
||||
// MiB peak even after the scoping fix above, which two concurrent workers cannot fit in a
|
||||
// 1 GiB container. Chaining 8000 -> 2048 -> 800 makes the second pass operate on 2048px
|
||||
// input, and the full-size buffer is already freed by the time it runs. Quality is not the
|
||||
// trade-off here: a staged Lanczos3 downscale to 800px is visually indistinguishable from
|
||||
// a single-step one (and is a standard technique for large ratios).
|
||||
display
|
||||
.resize(
|
||||
preview_max,
|
||||
preview_max,
|
||||
image::imageops::FilterType::Lanczos3,
|
||||
)
|
||||
.save_with_format(preview_path, image::ImageFormat::Jpeg)
|
||||
.context("failed to save preview")?;
|
||||
drop(display);
|
||||
|
||||
let pixels = u64::from(width) * u64::from(height);
|
||||
|
||||
// If the original is PNG, try lossless compression in place — but only when its pixel count
|
||||
// is inside the budget, and never for longer than OXIPNG_TIMEOUT. This is a best-effort size
|
||||
// saving: declining it costs disk, while attempting it unbounded cost the whole container.
|
||||
if mime_type == "image/png" {
|
||||
if pixels <= CompressionWorker::OXIPNG_MAX_PIXELS {
|
||||
let mut opts = oxipng::Options::from_preset(2);
|
||||
opts.timeout = Some(CompressionWorker::OXIPNG_TIMEOUT);
|
||||
let _ = oxipng::optimize(
|
||||
&oxipng::InFile::Path(original.to_path_buf()),
|
||||
&oxipng::OutFile::Path {
|
||||
path: None,
|
||||
preserve_attrs: true,
|
||||
},
|
||||
&opts,
|
||||
);
|
||||
} else {
|
||||
tracing::info!(
|
||||
%upload_id, pixels,
|
||||
"skipping oxipng: above the pixel budget; the original is stored as uploaded"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Peak resident set of THIS process, in bytes, from `/proc/self/status`.
|
||||
fn peak_rss_bytes() -> u64 {
|
||||
let status = std::fs::read_to_string("/proc/self/status").expect("procfs");
|
||||
let line = status
|
||||
.lines()
|
||||
.find(|l| l.starts_with("VmHWM:"))
|
||||
.expect("VmHWM");
|
||||
let kb: u64 = line
|
||||
.split_whitespace()
|
||||
.nth(1)
|
||||
.and_then(|v| v.parse().ok())
|
||||
.expect("VmHWM value");
|
||||
kb * 1024
|
||||
}
|
||||
|
||||
/// Reset the kernel's peak-RSS watermark so the measurement covers only what follows.
|
||||
/// Linux 4.0+; writing "5" to `clear_refs` resets `VmHWM` to the current RSS.
|
||||
fn reset_peak_rss() {
|
||||
let _ = std::fs::write("/proc/self/clear_refs", "5");
|
||||
}
|
||||
|
||||
/// The pixel gate has to sit below what the axis limits allow, or it can never fire.
|
||||
#[test]
|
||||
fn the_oxipng_gate_is_reachable_within_the_decode_limits() {
|
||||
const _: () = {
|
||||
// imaging::decode_limits permits 12_000 x 12_000 = 144 MP. A gate above that would
|
||||
// never skip anything.
|
||||
assert!(CompressionWorker::OXIPNG_MAX_PIXELS < 12_000 * 12_000);
|
||||
// ...and it must stay above a 48 MP camera, so real photos still get optimised.
|
||||
assert!(CompressionWorker::OXIPNG_MAX_PIXELS >= 8_000_000);
|
||||
};
|
||||
}
|
||||
|
||||
/// The heavy-image gate has to classify the two cases the way the sizing assumed:
|
||||
/// an ordinary phone photo must NOT serialise, and the giant must.
|
||||
#[test]
|
||||
fn the_heavy_gate_separates_a_phone_photo_from_a_giant() {
|
||||
let dir = std::env::temp_dir().join(format!("es-heavy-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
|
||||
// 12 MP, the shape of a default phone capture.
|
||||
let ordinary = dir.join("ordinary.jpg");
|
||||
image::RgbImage::new(4032, 3024).save(&ordinary).unwrap();
|
||||
let ordinary_peak = crate::services::imaging::estimated_processing_peak_bytes(
|
||||
&ordinary,
|
||||
CompressionWorker::DISPLAY_MAX_EDGE,
|
||||
)
|
||||
.expect("header readable");
|
||||
assert!(
|
||||
ordinary_peak <= CompressionWorker::HEAVY_IMAGE_BYTES,
|
||||
"a 12 MP photo estimated at {} MiB would serialise the common path",
|
||||
ordinary_peak / 1048576
|
||||
);
|
||||
|
||||
// The 64 MP RGBA case that measured ~516 MiB peak.
|
||||
let giant = dir.join("giant.png");
|
||||
image::RgbaImage::new(8000, 8000).save(&giant).unwrap();
|
||||
let giant_peak = crate::services::imaging::estimated_processing_peak_bytes(
|
||||
&giant,
|
||||
CompressionWorker::DISPLAY_MAX_EDGE,
|
||||
)
|
||||
.expect("header readable");
|
||||
assert!(
|
||||
giant_peak > CompressionWorker::HEAVY_IMAGE_BYTES,
|
||||
"an 8000x8000 RGBA original estimated at only {} MiB would be allowed to run \
|
||||
concurrently with another one — 2x its real ~516 MiB peak does not fit in 1 GiB",
|
||||
giant_peak / 1048576
|
||||
);
|
||||
// The estimate must also be in the right ballpark, not merely on the right side of the
|
||||
// threshold: 244 MiB decode + 262 MiB f32 resize intermediate.
|
||||
assert!(
|
||||
(400..700).contains(&(giant_peak / 1048576)),
|
||||
"estimate {} MiB is far from the measured ~516 MiB peak",
|
||||
giant_peak / 1048576
|
||||
);
|
||||
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
/// The OOM that took the container down, measured rather than argued.
|
||||
///
|
||||
/// An 8000x8000 RGBA PNG passes admission: 256,000,000 bytes is just under the 256 MiB
|
||||
/// `max_alloc`, and smooth content is a few MB on disk, far under any size cap. The old
|
||||
/// code kept that ~244 MiB decode alive across an unbounded, multi-threaded oxipng run and
|
||||
/// peaked at ~1250 MiB — inside a 1 GiB cgroup. Being SIGKILLed there is not a blip: the
|
||||
/// row was already committed, so the boot backfill replayed the identical workload on every
|
||||
/// restart.
|
||||
///
|
||||
/// `#[ignore]` because it allocates ~250 MiB and takes a few seconds. Run explicitly:
|
||||
/// cargo test --release oom -- --ignored --nocapture --test-threads=1
|
||||
/// It must run ALONE — `VmHWM` is per process, so a concurrent test would pollute it.
|
||||
#[test]
|
||||
#[ignore = "heavy: allocates ~250 MiB; run with --ignored --test-threads=1"]
|
||||
fn a_large_png_stays_far_below_the_container_limit() {
|
||||
const EDGE: u32 = 8_000;
|
||||
let dir = std::env::temp_dir().join(format!("es-oom-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let original = dir.join("big.png");
|
||||
|
||||
// Smooth gradient: ~244 MiB decoded, a couple of MB on disk. That gap is the whole
|
||||
// point — file size tells you nothing about what a PNG costs to process.
|
||||
{
|
||||
let mut buf = image::RgbaImage::new(EDGE, EDGE);
|
||||
for (x, y, px) in buf.enumerate_pixels_mut() {
|
||||
*px = image::Rgba([(x >> 5) as u8, (y >> 5) as u8, ((x + y) >> 6) as u8, 255]);
|
||||
}
|
||||
buf.save(&original).unwrap();
|
||||
}
|
||||
|
||||
// Everything above is fixture setup, not the code under test.
|
||||
reset_peak_rss();
|
||||
let before = peak_rss_bytes();
|
||||
|
||||
write_image_derivatives(
|
||||
Uuid::new_v4(),
|
||||
&original,
|
||||
"image/png",
|
||||
&dir.join("preview.jpg"),
|
||||
&dir.join("display.jpg"),
|
||||
)
|
||||
.expect("derivatives");
|
||||
|
||||
let peak = peak_rss_bytes();
|
||||
let on_disk = std::fs::metadata(&original).unwrap().len();
|
||||
eprintln!(
|
||||
"input {:.2} MiB on disk ({EDGE}x{EDGE}); peak RSS {:.0} MiB (was {:.0} MiB before)",
|
||||
on_disk as f64 / 1048576.0,
|
||||
peak as f64 / 1048576.0,
|
||||
before as f64 / 1048576.0
|
||||
);
|
||||
|
||||
assert!(dir.join("preview.jpg").exists() && dir.join("display.jpg").exists());
|
||||
// The container gets 1 GiB and runs two of these concurrently. 600 MiB is a generous
|
||||
// ceiling that the old code (~1250 MiB) could not have met.
|
||||
assert!(
|
||||
peak < 600 * 1024 * 1024,
|
||||
"peak RSS {} MiB — the decode is being held across oxipng again, or the pixel \
|
||||
gate stopped firing",
|
||||
peak / 1048576
|
||||
);
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
}
|
||||
|
||||
150
backend/src/services/config.rs
Normal file
150
backend/src/services/config.rs
Normal file
@@ -0,0 +1,150 @@
|
||||
//! Reads of the runtime-tunable `config` table, fronted by an in-memory cache.
|
||||
//!
|
||||
//! Each handler used to keep a small local copy of these helpers; consolidating them
|
||||
//! here means one place to add a parser, one place to mock for tests, and one place to
|
||||
//! find when a key changes. New keys do not require code changes — they're picked up
|
||||
//! the next time the cache reloads.
|
||||
//!
|
||||
//! ## Why a cache
|
||||
//!
|
||||
//! The `config` table is effectively static during an event, yet it was the busiest
|
||||
//! query in the system: every request re-read each key with its own `SELECT`
|
||||
//! (an upload touched it ~8 times). Against the small connection pool that was the
|
||||
//! throughput ceiling. [`ConfigCache`] loads the whole table once and serves reads
|
||||
//! from memory.
|
||||
//!
|
||||
//! ## Consistency contract
|
||||
//!
|
||||
//! Correctness comes from **synchronous invalidation on every write**, not from the
|
||||
//! TTL. The two runtime write paths — the admin `PATCH /admin/config` handler and the
|
||||
//! test-mode truncate/reseed — both call [`ConfigCache::invalidate`] after committing,
|
||||
//! so the *next* read reloads from the DB and sees the new value immediately. The
|
||||
//! [`RELOAD_TTL`] is only a safety net for out-of-band changes (e.g. a migration or a
|
||||
//! manual DB edit); it is deliberately short but never the primary mechanism.
|
||||
//!
|
||||
//! Values are read with a default fallback so the app still starts if a key is missing
|
||||
//! (e.g. during a migration window). Production seeds keys via migrations 005 and 009.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::sync::{Arc, RwLock};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use sqlx::PgPool;
|
||||
|
||||
/// How long a loaded snapshot is trusted before the next read reloads it. This is a
|
||||
/// backstop for out-of-band DB changes only — every in-process write invalidates the
|
||||
/// cache synchronously, so tests that PATCH-then-assert never depend on this expiring.
|
||||
const RELOAD_TTL: Duration = Duration::from_secs(30);
|
||||
|
||||
struct Snapshot {
|
||||
values: HashMap<String, String>,
|
||||
loaded_at: Instant,
|
||||
}
|
||||
|
||||
/// In-memory cache of the entire `config` table. Cheap to `clone` (shares the pool and
|
||||
/// the `Arc`), so it lives in `AppState` and every handler reads through it.
|
||||
#[derive(Clone)]
|
||||
pub struct ConfigCache {
|
||||
pool: PgPool,
|
||||
inner: Arc<RwLock<Option<Snapshot>>>,
|
||||
}
|
||||
|
||||
impl ConfigCache {
|
||||
pub fn new(pool: PgPool) -> Self {
|
||||
Self {
|
||||
pool,
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop the cached snapshot so the next read reloads the whole table from the DB.
|
||||
/// Call this after any write to the `config` table (admin PATCH, test reseed).
|
||||
pub fn invalidate(&self) {
|
||||
*self.inner.write().unwrap() = None;
|
||||
}
|
||||
|
||||
/// Return the fresh snapshot if one is loaded and still within [`RELOAD_TTL`].
|
||||
fn fresh_snapshot(&self) -> Option<HashMap<String, String>> {
|
||||
let guard = self.inner.read().unwrap();
|
||||
match guard.as_ref() {
|
||||
Some(snap) if snap.loaded_at.elapsed() < RELOAD_TTL => Some(snap.values.clone()),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Read one key, loading the whole table into the cache on a miss/expiry. On a DB
|
||||
/// error we return `None` (callers fall back to their default) without poisoning
|
||||
/// the cache.
|
||||
async fn get_raw(&self, key: &str) -> Option<String> {
|
||||
if let Some(values) = self.fresh_snapshot() {
|
||||
return values.get(key).cloned();
|
||||
}
|
||||
|
||||
// Cache miss or stale — reload the entire table in one query.
|
||||
let rows: Vec<(String, String)> = match sqlx::query_as::<_, (String, String)>(
|
||||
"SELECT key, value FROM config",
|
||||
)
|
||||
.fetch_all(&self.pool)
|
||||
.await
|
||||
{
|
||||
Ok(rows) => rows,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "config reload failed; using defaults for this read");
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
let values: HashMap<String, String> = rows.into_iter().collect();
|
||||
let result = values.get(key).cloned();
|
||||
*self.inner.write().unwrap() = Some(Snapshot {
|
||||
values,
|
||||
loaded_at: Instant::now(),
|
||||
});
|
||||
result
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn get_str(cache: &ConfigCache, key: &str, default: &str) -> String {
|
||||
cache
|
||||
.get_raw(key)
|
||||
.await
|
||||
.unwrap_or_else(|| default.to_string())
|
||||
}
|
||||
|
||||
pub async fn get_i64(cache: &ConfigCache, key: &str, default: i64) -> i64 {
|
||||
cache
|
||||
.get_raw(key)
|
||||
.await
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
pub async fn get_usize(cache: &ConfigCache, key: &str, default: usize) -> usize {
|
||||
cache
|
||||
.get_raw(key)
|
||||
.await
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
pub async fn get_f64(cache: &ConfigCache, key: &str, default: f64) -> f64 {
|
||||
cache
|
||||
.get_raw(key)
|
||||
.await
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
/// Parses common truthy spellings used by both the migration seeds and the admin form.
|
||||
/// Accepts `true/false`, `1/0`, `yes/no`, `on/off` — case-insensitive. Anything else
|
||||
/// returns `default`.
|
||||
pub async fn get_bool(cache: &ConfigCache, key: &str, default: bool) -> bool {
|
||||
let Some(raw) = cache.get_raw(key).await else {
|
||||
return default;
|
||||
};
|
||||
match raw.trim().to_ascii_lowercase().as_str() {
|
||||
"true" | "1" | "yes" | "on" => true,
|
||||
"false" | "0" | "no" | "off" => false,
|
||||
_ => default,
|
||||
}
|
||||
}
|
||||
169
backend/src/services/disk.rs
Normal file
169
backend/src/services/disk.rs
Normal file
@@ -0,0 +1,169 @@
|
||||
//! Cached view of the filesystem backing the media directory.
|
||||
//!
|
||||
//! Free/total disk space is needed on two hot paths — the per-user storage quota
|
||||
//! (checked on every upload *and* every `GET /me/quota` poll) and the admin stats
|
||||
//! endpoint. Reading it means `sysinfo::Disks::new_with_refreshed_list()`, which stats
|
||||
//! every mounted filesystem; doing that per request is wasteful for a number that
|
||||
//! barely moves. [`DiskCache`] refreshes it at most once per [`TTL`] and serves the
|
||||
//! rest from memory.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, RwLock};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// How long a disk reading is trusted before the next call re-stats the filesystem.
|
||||
const TTL: Duration = Duration::from_secs(15);
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct DiskInfo {
|
||||
pub total: u64,
|
||||
pub free: u64,
|
||||
}
|
||||
|
||||
/// Cheap-to-clone cache of the media filesystem's total/free bytes. Lives in
|
||||
/// `AppState`.
|
||||
#[derive(Clone)]
|
||||
pub struct DiskCache {
|
||||
inner: Arc<RwLock<Option<(PathBuf, DiskInfo, Instant)>>>,
|
||||
}
|
||||
|
||||
impl DiskCache {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop the cached reading so the next `snapshot()` re-measures the filesystem.
|
||||
///
|
||||
/// Used by the e2e TRUNCATE endpoint. Truncating deletes every uploaded file, which materially
|
||||
/// changes free space — but the cached reading survives for up to the TTL, so the next test can
|
||||
/// compute a quota from the PREVIOUS test's disk. That matters now that the quota tests steer
|
||||
/// the per-user limit off `free_disk_bytes`: a stale reading makes the limit wrong and the test
|
||||
/// flaky, for reasons that have nothing to do with the code under test.
|
||||
pub fn invalidate(&self) {
|
||||
*self.inner.write().unwrap() = None;
|
||||
}
|
||||
|
||||
/// Cached `(total, free)` bytes for the filesystem that holds `media_path`.
|
||||
///
|
||||
/// Returns `None` when the mount can't be resolved — callers MUST treat that as
|
||||
/// "unknown", never "zero free". (The quota path in particular fails *open* on
|
||||
/// `None`: enforcing a 0-byte limit would lock every user out of uploading.)
|
||||
/// Cached free-space reading for `path`.
|
||||
///
|
||||
/// The cache is keyed BY PATH. It used to hold a single slot and ignore its argument on a hit,
|
||||
/// so it would happily return the media volume's numbers for any other path within the TTL.
|
||||
/// That was invisible only because every caller happened to pass `media_path` — the first
|
||||
/// caller to ask about a different volume (e.g. the exports volume, which is a separate mount)
|
||||
/// would have silently got the wrong filesystem's free space.
|
||||
pub fn snapshot(&self, path: &Path) -> Option<DiskInfo> {
|
||||
if let Some((cached_path, info, at)) = self.inner.read().unwrap().as_ref()
|
||||
&& cached_path == path
|
||||
&& at.elapsed() < TTL
|
||||
{
|
||||
return Some(*info);
|
||||
}
|
||||
let info = read_disk_for_path(path)?;
|
||||
*self.inner.write().unwrap() = Some((path.to_path_buf(), info, Instant::now()));
|
||||
Some(info)
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for DiskCache {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
/// UNCACHED free-space reading for the filesystem backing `path`.
|
||||
///
|
||||
/// Deliberately bypasses [`DiskCache`]. The cache exists for the quota poll, where a 15s-stale
|
||||
/// number is fine because it is only ever advisory. The export preflight is the opposite case: it
|
||||
/// decides whether to start writing a multi-GB archive, and the sibling export worker running
|
||||
/// concurrently can move free space by tens of gigabytes well inside the TTL. A stale reading there
|
||||
/// would authorise exactly the write that fills the disk.
|
||||
pub fn free_bytes(path: &Path) -> Option<u64> {
|
||||
read_disk_for_path(path).map(|d| d.free)
|
||||
}
|
||||
|
||||
/// Resolve the filesystem backing `media_path` and read its total/free bytes.
|
||||
///
|
||||
/// Snapshots the mount table via `sysinfo`, then delegates the selection to the pure
|
||||
/// [`select_disk`] so the (fiddly, edge-case-prone) matching logic is unit-testable
|
||||
/// without touching the real filesystem.
|
||||
fn read_disk_for_path(media_path: &Path) -> Option<DiskInfo> {
|
||||
let disks = sysinfo::Disks::new_with_refreshed_list();
|
||||
let mounts: Vec<(String, u64, u64)> = disks
|
||||
.iter()
|
||||
.map(|d| {
|
||||
(
|
||||
d.mount_point().to_string_lossy().to_string(),
|
||||
d.total_space(),
|
||||
d.available_space(),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
select_disk(&mounts, &media_path.to_string_lossy())
|
||||
}
|
||||
|
||||
/// Pick the filesystem for `media_path` from a `(mount_point, total, free)` table.
|
||||
///
|
||||
/// Chooses the **longest** mount point that is a prefix of `media_path` (the most
|
||||
/// specific filesystem) rather than the first match — otherwise the root `/` mount,
|
||||
/// which prefixes every absolute path, could shadow a dedicated `/media` volume. Falls
|
||||
/// back to `/` when nothing prefixes the path (e.g. a relative media path), and to
|
||||
/// `None` when even that is absent — the caller treats `None` as "unknown" and fails
|
||||
/// open on quota.
|
||||
fn select_disk(mounts: &[(String, u64, u64)], media_path: &str) -> Option<DiskInfo> {
|
||||
mounts
|
||||
.iter()
|
||||
.filter(|(mp, _, _)| media_path.starts_with(mp.as_str()))
|
||||
.max_by_key(|(mp, _, _)| mp.len())
|
||||
.or_else(|| mounts.iter().find(|(mp, _, _)| mp == "/"))
|
||||
.map(|(_, total, free)| DiskInfo {
|
||||
total: *total,
|
||||
free: *free,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::select_disk;
|
||||
|
||||
#[test]
|
||||
fn picks_longest_matching_mount() {
|
||||
// Both "/" and "/media" prefix the path; the dedicated volume must win.
|
||||
let mounts = vec![("/".to_string(), 100, 40), ("/media".to_string(), 200, 150)];
|
||||
let d = select_disk(&mounts, "/media/originals/x.jpg").unwrap();
|
||||
assert_eq!((d.total, d.free), (200, 150));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn falls_back_to_root_when_no_specific_mount_matches() {
|
||||
let mounts = vec![("/".to_string(), 100, 40), ("/media".to_string(), 200, 150)];
|
||||
// "/var/lib" is only prefixed by "/".
|
||||
let d = select_disk(&mounts, "/var/lib/data").unwrap();
|
||||
assert_eq!((d.total, d.free), (100, 40));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_path_uses_root_fallback() {
|
||||
let mounts = vec![("/".to_string(), 100, 40)];
|
||||
// A relative path prefixes nothing, so the explicit "/" fallback applies.
|
||||
let d = select_disk(&mounts, "media/originals").unwrap();
|
||||
assert_eq!((d.total, d.free), (100, 40));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn none_when_no_mount_matches_and_no_root() {
|
||||
// No "/" present and nothing prefixes the relative path → unknown (fail-open).
|
||||
let mounts = vec![("/data".to_string(), 100, 40)];
|
||||
assert!(select_disk(&mounts, "relative/path").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn none_on_empty_mount_table() {
|
||||
assert!(select_disk(&[], "/media/x").is_none());
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
297
backend/src/services/imaging.rs
Normal file
297
backend/src/services/imaging.rs
Normal file
@@ -0,0 +1,297 @@
|
||||
//! Shared image decoding.
|
||||
//!
|
||||
//! Exists so there is exactly ONE way to turn a file on disk into a `DynamicImage` in this
|
||||
//! codebase. Two properties have to hold everywhere an image is decoded, and both were
|
||||
//! previously re-derived per call site — which is how they drifted apart:
|
||||
//!
|
||||
//! - **EXIF orientation must be applied.** Phones do not rotate sensor data; they record how
|
||||
//! the camera was held in a tag and store the pixels as shot. `image::open` and
|
||||
//! `ImageReader::decode` both hand back the raw pixels and ignore that tag, and re-encoding
|
||||
//! to JPEG writes no EXIF, so the derivative is permanently sideways while the untouched
|
||||
//! original still renders upright. The compression worker was fixed; the export worker was
|
||||
//! not, so every portrait photo came out sideways in the keepsake's HTML viewer.
|
||||
//! - **Decode limits must be set.** The upload body cap bounds the file on disk, but a small
|
||||
//! file can decode to enormous dimensions (a ~1 MB image expanding to 50k×50k px), OOM-ing
|
||||
//! the box. `image::open` applies NO limits at all, so the export path was also decoding
|
||||
//! arbitrary user-supplied images unbounded.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use image::{DynamicImage, ImageDecoder};
|
||||
use std::path::Path;
|
||||
|
||||
/// Bounds for any decode of user-supplied image data. The per-axis cap covers any real phone
|
||||
/// photo; `max_alloc` bounds the decoded buffer — but only because `decode_oriented` reserves
|
||||
/// against it explicitly, see there.
|
||||
///
|
||||
/// Sized against the deployment: the app container is capped at 1 GiB and the compression
|
||||
/// worker runs `compression_concurrency` decodes at once (default 2), so 256 MiB per decode
|
||||
/// leaves headroom for the resize buffers and the runtime.
|
||||
fn decode_limits() -> image::Limits {
|
||||
let mut limits = image::Limits::default();
|
||||
limits.max_image_width = Some(12_000);
|
||||
limits.max_image_height = Some(12_000);
|
||||
limits.max_alloc = Some(256 * 1024 * 1024);
|
||||
limits
|
||||
}
|
||||
|
||||
/// True when re-running the exact same work on the exact same bytes cannot possibly
|
||||
/// succeed, so retrying only burns wall-clock and log noise.
|
||||
///
|
||||
/// Deliberately narrow. Only the `ImageError` variants that are a property of the *input*
|
||||
/// count: the file will not shrink, gain codec support, or un-corrupt itself between
|
||||
/// attempts. `IoError` is excluded on purpose — an ENOSPC while writing a derivative, or
|
||||
/// EMFILE under load, is exactly the transient case the retry exists for.
|
||||
pub fn is_permanent_image_error(err: &anyhow::Error) -> bool {
|
||||
err.chain().any(|cause| {
|
||||
matches!(
|
||||
cause.downcast_ref::<image::ImageError>(),
|
||||
Some(
|
||||
image::ImageError::Limits(_)
|
||||
| image::ImageError::Unsupported(_)
|
||||
| image::ImageError::Decoding(_)
|
||||
)
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
/// Build a decoder for `path` with the budget enforced, WITHOUT reading any pixels.
|
||||
///
|
||||
/// Single source of truth for "may this image be decoded at all": both the upload
|
||||
/// admission check and the compression worker go through here, so they cannot disagree
|
||||
/// about what is acceptable.
|
||||
fn decoder_within_budget(path: &Path) -> Result<impl image::ImageDecoder> {
|
||||
let mut reader = image::ImageReader::open(path)
|
||||
.context("failed to open image")?
|
||||
.with_guessed_format()
|
||||
.context("failed to read image header")?;
|
||||
let mut limits = decode_limits();
|
||||
reader.limits(limits.clone());
|
||||
|
||||
// We need `into_decoder` rather than `decode()` to read the EXIF orientation tag before
|
||||
// the pixels are consumed. But the two are NOT equivalent on safety: `decode()` performs
|
||||
//
|
||||
// limits.reserve(decoder.total_bytes())?;
|
||||
//
|
||||
// between building the decoder and reading the image, and `into_decoder()` skips it (the
|
||||
// crate's own FIXME concedes `from_decoder` doesn't compensate). Nothing else enforces
|
||||
// `max_alloc` — the JPEG decoder's `set_limits` only checks support and dimensions — so
|
||||
// without the line below the budget is inert and the ONLY bound is the per-axis cap. That
|
||||
// leaves 12000x12000 decodable at 412 MiB, and two concurrent at 824 MiB against a 1 GiB
|
||||
// container. Re-add it, exactly as `decode()` does.
|
||||
let mut decoder = reader.into_decoder().context("failed to decode image")?;
|
||||
limits
|
||||
.reserve(decoder.total_bytes())
|
||||
.context("image too large to decode within the memory budget")?;
|
||||
decoder
|
||||
.set_limits(limits)
|
||||
.context("image too large to decode within the memory budget")?;
|
||||
Ok(decoder)
|
||||
}
|
||||
|
||||
/// Rough peak heap an image will cost to turn into derivatives, read from the HEADER only —
|
||||
/// no pixels are decoded. `None` when the header can't be read or the image is over budget
|
||||
/// (the caller is about to fail on it anyway).
|
||||
///
|
||||
/// Two terms, and the second is the one that surprises:
|
||||
///
|
||||
/// - the decoded buffer, `width * height * channels`; and
|
||||
/// - the resize intermediate. `image`'s Lanczos3 path accumulates in `f32`, so the buffer
|
||||
/// between the horizontal and vertical passes is `new_width * old_height * 4 channels * 4
|
||||
/// bytes` — 16 bytes per pixel-row-slot, not the 4 the output uses. For an 8000x8000
|
||||
/// original that is 262 MiB on top of a 244 MiB decode, measured. It is bigger than the
|
||||
/// decode for any tall image, which is why "the decode is bounded by max_alloc" was never
|
||||
/// the whole story.
|
||||
///
|
||||
/// Used to decide whether an image is heavy enough to need exclusive use of the box's memory
|
||||
/// headroom, NOT to reject anything.
|
||||
pub fn estimated_processing_peak_bytes(path: &Path, display_edge: u32) -> Option<u64> {
|
||||
let decoder = decoder_within_budget(path).ok()?;
|
||||
let (width, height) = decoder.dimensions();
|
||||
let decoded = decoder.total_bytes();
|
||||
|
||||
// Aspect-preserving fit into `display_edge`, matching DynamicImage::resize. No downscale
|
||||
// means no intermediate at all.
|
||||
let intermediate = if width > display_edge || height > display_edge {
|
||||
let ratio = f64::from(display_edge) / f64::from(width.max(height));
|
||||
let new_width = (f64::from(width) * ratio).round().max(1.0) as u64;
|
||||
new_width * u64::from(height) * 16
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
Some(decoded.saturating_add(intermediate))
|
||||
}
|
||||
|
||||
/// Megapixels an image would decode to, or `None` if its header can't be read. Used only
|
||||
/// to put a concrete number in the message the guest sees.
|
||||
pub fn megapixels(path: &Path) -> Option<f64> {
|
||||
let reader = image::ImageReader::open(path)
|
||||
.ok()?
|
||||
.with_guessed_format()
|
||||
.ok()?;
|
||||
let (w, h) = reader.into_dimensions().ok()?;
|
||||
Some(f64::from(w) * f64::from(h) / 1_000_000.0)
|
||||
}
|
||||
|
||||
/// True when an image cannot be decoded specifically because it would exceed the memory
|
||||
/// budget — read from the header, no pixels touched.
|
||||
///
|
||||
/// Called at upload admission so a guest who sends a 100 MP photo is told at the door, with
|
||||
/// a reason they can act on, instead of the upload being accepted with a 201 and then
|
||||
/// silently soft-deleted minutes later when the worker gives up on it.
|
||||
///
|
||||
/// Deliberately narrow: ONLY the budget. A corrupt, truncated or unsupported file also
|
||||
/// fails to build a decoder, but rejecting those here would change a contract the
|
||||
/// adversarial suite pins on purpose — acceptance follows the magic bytes, and a payload
|
||||
/// with a valid JPEG header is accepted regardless of what follows it. Those go to the
|
||||
/// compression worker as before, which handles them gracefully and (since the retry
|
||||
/// classifier) no longer burns backoff on them.
|
||||
pub fn exceeds_decode_budget(path: &Path) -> bool {
|
||||
match decoder_within_budget(path) {
|
||||
Ok(_) => false,
|
||||
Err(e) => e.chain().any(|cause| {
|
||||
matches!(
|
||||
cause.downcast_ref::<image::ImageError>(),
|
||||
Some(image::ImageError::Limits(_))
|
||||
)
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode an image from disk with decompression-bomb limits applied and its EXIF
|
||||
/// orientation baked into the pixels.
|
||||
///
|
||||
/// Blocking — call inside `spawn_blocking`.
|
||||
pub fn decode_oriented(path: &Path) -> Result<DynamicImage> {
|
||||
let mut decoder = decoder_within_budget(path)?;
|
||||
|
||||
// Cheap, and it happens BEFORE any pixels are read: an oversized image costs a header
|
||||
// parse, not an allocation.
|
||||
let orientation = decoder
|
||||
.orientation()
|
||||
.unwrap_or(image::metadata::Orientation::NoTransforms);
|
||||
let mut img = DynamicImage::from_decoder(decoder).context("failed to decode image")?;
|
||||
img.apply_orientation(orientation);
|
||||
Ok(img)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Shared with the e2e suite rather than duplicating 568 KiB of binary: the same file
|
||||
/// drives `02-upload/oversized-image` so both layers assert on one artefact.
|
||||
const HUGE: &str = concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/../e2e/fixtures/media/huge-99mp.jpg"
|
||||
);
|
||||
|
||||
#[test]
|
||||
fn rejects_an_image_that_would_blow_the_allocation_budget() {
|
||||
// 11000x9000 = 99 MP. Deliberately UNDER the 12000px per-axis cap, so the axis check
|
||||
// cannot reject it — the allocation budget is the only thing that can, which is
|
||||
// exactly what makes this a regression test rather than a restatement of the axis cap.
|
||||
// 283 MiB decoded as RGB8 against a 256 MiB budget, from 568 KiB on disk.
|
||||
//
|
||||
// This failed before the guard was restored: `ImageReader::decode` performs
|
||||
// `limits.reserve(decoder.total_bytes())`, and `into_decoder()` — which we need for
|
||||
// the EXIF tag — skips it, so `max_alloc` was inert and this decoded happily.
|
||||
// Map the Ok arm to its dimensions first: on failure `expect_err` Debug-prints the
|
||||
// value, and Debug on a DynamicImage dumps every pixel — 283 MiB of output.
|
||||
let err = decode_oriented(Path::new(HUGE))
|
||||
.map(|img| (img.width(), img.height()))
|
||||
.expect_err("a 99 MP image must be refused, not allocated");
|
||||
let msg = format!("{err:#}");
|
||||
assert!(
|
||||
msg.to_lowercase().contains("limit") || msg.to_lowercase().contains("memory"),
|
||||
"expected a limits error, got: {msg}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_oversized_image_is_a_permanent_failure() {
|
||||
// The retry loop must not burn 2s + 4s of backoff on this: the file will not shrink
|
||||
// between attempts, so all three attempts reach the identical conclusion.
|
||||
let err = decode_oriented(Path::new(HUGE))
|
||||
.map(|img| (img.width(), img.height()))
|
||||
.expect_err("fixture must exceed the budget");
|
||||
assert!(
|
||||
is_permanent_image_error(&err),
|
||||
"a Limits error can never succeed on retry: {err:#}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_plain_io_error_is_not_permanent() {
|
||||
// The mirror that keeps the classifier honest. ENOSPC while writing a derivative, or
|
||||
// EMFILE under load, is exactly what the retry exists for — misclassifying those as
|
||||
// permanent would turn a transient blip back into the data loss round 1 fixed.
|
||||
let err = decode_oriented(Path::new("/nonexistent/definitely-not-here.jpg"))
|
||||
.map(|img| (img.width(), img.height()))
|
||||
.expect_err("a missing file must error");
|
||||
assert!(
|
||||
!is_permanent_image_error(&err),
|
||||
"an IO error must stay retryable: {err:#}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn admission_rejects_only_the_over_budget_case() {
|
||||
// Admission and processing must agree about SIZE — a photo accepted at the door and
|
||||
// then rejected by the worker for being too big is the failure this pair prevents.
|
||||
assert!(
|
||||
exceeds_decode_budget(Path::new(HUGE)),
|
||||
"admission must reject what the decoder rejects for size"
|
||||
);
|
||||
let ordinary = concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/../e2e/fixtures/media/portrait-exif6.jpg"
|
||||
);
|
||||
assert!(
|
||||
!exceeds_decode_budget(Path::new(ordinary)),
|
||||
"admission must accept an ordinary photo"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn admission_does_not_reject_a_merely_undecodable_file() {
|
||||
// The narrowing that keeps the adversarial contract intact: a payload with valid
|
||||
// JPEG magic bytes and nothing behind them cannot be decoded, but acceptance follows
|
||||
// the magic bytes by design (07-adversarial/file-upload-attacks). It is the worker's
|
||||
// job to fail it, not admission's — admission is only the resource guard.
|
||||
let dir = std::env::temp_dir().join("eventsnap-imaging-test");
|
||||
std::fs::create_dir_all(&dir).expect("tmp dir");
|
||||
let stub = dir.join("magic-only.jpg");
|
||||
let mut bytes = vec![0u8; 1024];
|
||||
bytes[..3].copy_from_slice(&[0xFF, 0xD8, 0xFF]);
|
||||
std::fs::write(&stub, &bytes).expect("write stub");
|
||||
|
||||
assert!(
|
||||
!exceeds_decode_budget(&stub),
|
||||
"a corrupt file is not an over-budget file"
|
||||
);
|
||||
assert!(
|
||||
decode_oriented(&stub)
|
||||
.map(|i| (i.width(), i.height()))
|
||||
.is_err(),
|
||||
"...but it must still fail in the worker"
|
||||
);
|
||||
let _ = std::fs::remove_file(&stub);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn still_decodes_an_ordinary_photo_and_applies_orientation() {
|
||||
// The guard must not have become a blanket refusal. This fixture is 40x20 stored with
|
||||
// EXIF Orientation=6, so a correct decode returns it rotated to 20x40 portrait.
|
||||
let path = concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/../e2e/fixtures/media/portrait-exif6.jpg"
|
||||
);
|
||||
let img = decode_oriented(Path::new(path)).expect("an ordinary photo must decode");
|
||||
assert_eq!(
|
||||
(img.width(), img.height()),
|
||||
(20, 40),
|
||||
"EXIF orientation must still be applied after restoring the guard"
|
||||
);
|
||||
}
|
||||
}
|
||||
398
backend/src/services/maintenance.rs
Normal file
398
backend/src/services/maintenance.rs
Normal file
@@ -0,0 +1,398 @@
|
||||
//! Startup recovery + periodic background hygiene.
|
||||
//!
|
||||
//! Two responsibilities:
|
||||
//!
|
||||
//! 1. **Startup sweep** — when the server boots, fix rows left in an "in-progress"
|
||||
//! state by the previous (possibly crashed) instance. Compression and export jobs
|
||||
//! each leave a status row when they begin; if the process is killed mid-run, that
|
||||
//! row stays `'processing'` / `'running'` forever, blocking re-tries and leaving
|
||||
//! users staring at a spinner. Resetting them on startup recovers gracefully.
|
||||
//!
|
||||
//! 2. **Periodic tasks** — pruning that should happen "every hour" rather than per
|
||||
//! request: expired sessions (otherwise the table grows unboundedly), the
|
||||
//! rate-limiter's in-memory windows (so keys for IPs that left long ago don't
|
||||
//! accumulate), and the media of soft-deleted uploads — both the ones whose compression
|
||||
//! permanently failed and the ones a guest or host deliberately removed — which are
|
||||
//! retained for a recovery window and then reclaimed.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::Duration;
|
||||
|
||||
use sqlx::PgPool;
|
||||
|
||||
use crate::services::rate_limiter::RateLimiter;
|
||||
use crate::services::sse_tickets::SseTicketStore;
|
||||
|
||||
/// How long a permanently-failed upload's original is kept on disk before it is
|
||||
/// reclaimed.
|
||||
///
|
||||
/// The compression worker stops deleting originals on failure — a transient error must
|
||||
/// never destroy the guest's only copy of a photo they can't retake. But the row is
|
||||
/// soft-deleted and the uploader's quota IS refunded, so without a sweep those bytes are
|
||||
/// invisible, unowned, and free: a reproducible codec failure lets one guest accumulate
|
||||
/// orphans at no personal cost, and because `active_uploaders` counts only users with
|
||||
/// non-deleted uploads, dropping out of that count actually RAISES everyone's per-user
|
||||
/// ceiling while the disk gets fuller.
|
||||
///
|
||||
/// Two weeks is comfortably longer than any single event, so an operator investigating a
|
||||
/// failed upload still has the file, while the leak stays bounded.
|
||||
const FAILED_ORIGINAL_RETENTION_DAYS: i64 = 14;
|
||||
|
||||
/// How long a DELIBERATELY deleted upload's files are kept before they are reclaimed.
|
||||
///
|
||||
/// The same leak, reached by the ordinary path rather than the exceptional one.
|
||||
/// `soft_delete_in_event` stamps `deleted_at` and refunds `total_upload_bytes`, but nothing ever
|
||||
/// removed the bytes — so the quota stopped bounding the disk. Upload 500 MB, delete, quota is back
|
||||
/// to zero, upload another 500 MB: not an attack, just a guest curating their camera roll, which is
|
||||
/// what people do. The host then sees guests hitting "Du hast dein Upload-Limit erreicht" while the
|
||||
/// admin widget shows a disk full of files no upload row points at, and the quota message is
|
||||
/// actively misleading because the space really is gone — just not to anyone the accounting can
|
||||
/// name.
|
||||
///
|
||||
/// Much shorter than the failure window on purpose. Fourteen days outlives the whole event, so a
|
||||
/// deliberate delete would never reclaim anything while it mattered. A day still gives an operator
|
||||
/// a recovery window for a mis-tap.
|
||||
///
|
||||
/// NOTE what this does NOT do: within the window the bytes are still spent and still unaccounted,
|
||||
/// so a guest deleting and re-uploading through an eight-hour event can outrun the sweep. Bounding
|
||||
/// that would mean holding the quota until the file is actually reclaimed rather than refunding at
|
||||
/// `deleted_at` — a deliberate trade, and the reason the low-disk warning exists.
|
||||
const DELETED_UPLOAD_RETENTION_HOURS: i64 = 24;
|
||||
|
||||
/// Reset rows left in flight by a previous crashed instance. Run once on startup,
|
||||
/// before the HTTP server starts taking requests, so users never observe the
|
||||
/// half-state.
|
||||
pub async fn startup_recovery(pool: &PgPool) {
|
||||
// Uploads whose preview generation was interrupted. Marking them 'failed' is
|
||||
// safer than re-queueing — the original file is still on disk, the user can
|
||||
// delete + re-upload if they care, and we avoid double-processing risk.
|
||||
match sqlx::query(
|
||||
"UPDATE upload SET compression_status = 'failed'
|
||||
WHERE compression_status = 'processing'",
|
||||
)
|
||||
.execute(pool)
|
||||
.await
|
||||
{
|
||||
Ok(r) if r.rows_affected() > 0 => {
|
||||
tracing::warn!(
|
||||
"startup recovery: reset {} stuck upload(s) from 'processing' to 'failed'",
|
||||
r.rows_affected()
|
||||
);
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(e) => tracing::error!("startup recovery: failed to sweep uploads: {e:#}"),
|
||||
}
|
||||
|
||||
// Export jobs interrupted mid-run are marked 'failed' here so they aren't left
|
||||
// 'running' forever. The host CANNOT re-trigger a released export (release_gallery
|
||||
// rejects an already-released event), so `export::recover_exports` re-spawns these
|
||||
// failed-but-released jobs from `main` once `AppState` exists (it needs the media
|
||||
// paths + SSE sender this fn doesn't have).
|
||||
//
|
||||
// SINGLE-INSTANCE ASSUMPTION: this sweep is unscoped — it reaps every `running` row, not just
|
||||
// this process's. That is correct for the supported deployment (one app container; see
|
||||
// docker-compose.yml), where no other process can own a job at boot. If EventSnap is ever run
|
||||
// with more than one replica, a booting replica would mark a peer's live workers `failed`.
|
||||
// This is NOT merely wasteful — do not assume the epoch model contains it. Recovery re-arms the
|
||||
// job at the SAME epoch, so the reaped-but-still-alive worker and the fresh one would BOTH hold
|
||||
// that epoch; the guards carry no owner/lease token, so they share the same artifact paths and
|
||||
// either can win the other's `finalize_job`. That can corrupt or delete the keepsake. Running
|
||||
// more than one replica therefore requires an owner/lease/heartbeat on export_job first. That
|
||||
// machinery is not worth building for a single-container app — so this is a documented
|
||||
// CONSTRAINT, not a hidden assumption: do not scale this service horizontally as-is.
|
||||
match sqlx::query(
|
||||
"UPDATE export_job
|
||||
SET status = 'failed',
|
||||
error_message = COALESCE(error_message, 'Server-Neustart während des Exports')
|
||||
WHERE status = 'running'",
|
||||
)
|
||||
.execute(pool)
|
||||
.await
|
||||
{
|
||||
Ok(r) if r.rows_affected() > 0 => {
|
||||
tracing::warn!(
|
||||
"startup recovery: reset {} stuck export job(s) from 'running' to 'failed'",
|
||||
r.rows_affected()
|
||||
);
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(e) => tracing::error!("startup recovery: failed to sweep export_job: {e:#}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// How long a file in `originals/` may exist without a database row before it is treated as
|
||||
/// abandoned.
|
||||
///
|
||||
/// This window is the ONLY thing making the sweep safe, because the upload handler renames the
|
||||
/// temp file into its final path BEFORE committing the row: for a short moment a perfectly
|
||||
/// healthy upload legitimately looks exactly like an orphan. Six hours is far beyond any live
|
||||
/// request (a 576 MiB body over a bad venue uplink is minutes, and the request itself is bounded
|
||||
/// by the reverse proxy) while still reclaiming the leak inside a single event.
|
||||
///
|
||||
/// DO NOT SHORTEN THIS to make a test faster — a value below the longest possible in-flight
|
||||
/// upload deletes photos out from under the request that is committing them.
|
||||
const ORPHAN_UPLOAD_RETENTION_HOURS: u64 = 6;
|
||||
|
||||
/// Spawns a background task that periodically:
|
||||
/// - deletes session rows whose `expires_at` is more than a day in the past
|
||||
/// - prunes the in-memory rate-limiter HashMap of empty windows
|
||||
/// - drops expired SSE tickets (30s TTL but the map keeps the slot until pruned)
|
||||
///
|
||||
/// Cadence is 1h — fine for both jobs at our scale.
|
||||
pub fn spawn_periodic_tasks(
|
||||
pool: PgPool,
|
||||
rate_limiter: RateLimiter,
|
||||
sse_tickets: SseTicketStore,
|
||||
media_path: PathBuf,
|
||||
) {
|
||||
tokio::spawn(async move {
|
||||
let mut tick = tokio::time::interval(Duration::from_secs(3600));
|
||||
// Fire the first tick immediately, then hourly.
|
||||
tick.tick().await;
|
||||
loop {
|
||||
tick.tick().await;
|
||||
cleanup_sessions(&pool).await;
|
||||
cleanup_deleted_media(&pool, &media_path).await;
|
||||
sweep_orphan_originals(&pool, &media_path).await;
|
||||
rate_limiter.prune();
|
||||
sse_tickets.prune();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Reclaim the media of soft-deleted uploads once they are past their retention window.
|
||||
///
|
||||
/// ONLY ever touches rows with `deleted_at IS NOT NULL`, so it can never reach a live upload. Two
|
||||
/// classes, two windows, because the two deletes mean different things:
|
||||
///
|
||||
/// - a compression failure the guest didn't ask for and may want investigated —
|
||||
/// [`FAILED_ORIGINAL_RETENTION_DAYS`];
|
||||
/// - a deliberate removal by the guest or the host — [`DELETED_UPLOAD_RETENTION_HOURS`].
|
||||
///
|
||||
/// ALL FOUR paths are reclaimed, not just the original. The previous version cleared
|
||||
/// `original_path` alone, which was right for its only case (a failed compression produces no
|
||||
/// derivatives) but wrong the moment the sweep reaches a successfully processed upload: preview,
|
||||
/// display and thumbnail are each a separate file on disk, none of them counted in
|
||||
/// `original_size_bytes`, and nothing else ever removed them.
|
||||
///
|
||||
/// Every column is cleared in the same pass, which makes the sweep idempotent and stops a later run
|
||||
/// re-reporting files that are already gone. The ROW is kept: it is the audit trail, it costs a few
|
||||
/// hundred bytes, and `backfill_stale_derivatives` is guarded on `deleted_at IS NULL` so a nulled
|
||||
/// `preview_path` can never make it regenerate what was just reclaimed.
|
||||
async fn cleanup_deleted_media(pool: &PgPool, media_path: &std::path::Path) {
|
||||
type Row = (
|
||||
uuid::Uuid,
|
||||
String,
|
||||
Option<String>,
|
||||
Option<String>,
|
||||
Option<String>,
|
||||
);
|
||||
let rows = sqlx::query_as::<_, Row>(
|
||||
"SELECT id, original_path, preview_path, display_path, thumbnail_path FROM upload
|
||||
WHERE deleted_at IS NOT NULL
|
||||
AND CASE WHEN compression_status = 'failed'
|
||||
THEN deleted_at < NOW() - ($1 || ' days')::interval
|
||||
ELSE deleted_at < NOW() - ($2 || ' hours')::interval
|
||||
END
|
||||
AND (original_path <> '' OR preview_path IS NOT NULL
|
||||
OR display_path IS NOT NULL OR thumbnail_path IS NOT NULL)",
|
||||
)
|
||||
.bind(FAILED_ORIGINAL_RETENTION_DAYS.to_string())
|
||||
.bind(DELETED_UPLOAD_RETENTION_HOURS.to_string())
|
||||
.fetch_all(pool)
|
||||
.await;
|
||||
|
||||
let rows = match rows {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "deleted-media sweep query failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
if rows.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut reclaimed = 0usize;
|
||||
for (id, original, preview, display, thumbnail) in rows {
|
||||
let paths: Vec<String> = std::iter::once(original)
|
||||
.filter(|p| !p.is_empty())
|
||||
.chain([preview, display, thumbnail].into_iter().flatten())
|
||||
.collect();
|
||||
|
||||
// All-or-nothing per row: the columns are only cleared once every file for that upload is
|
||||
// gone. Clearing after a partial success would strand the survivors with nothing pointing
|
||||
// at them — the same unowned-bytes state this sweep exists to drain.
|
||||
let mut all_gone = true;
|
||||
for rel in &paths {
|
||||
let absolute = media_path.join(rel);
|
||||
match tokio::fs::remove_file(&absolute).await {
|
||||
Ok(()) => reclaimed += 1,
|
||||
// Already gone (manual cleanup, restored backup) — still counts as reclaimed for
|
||||
// the purpose of clearing the columns, or the row is re-selected every hour forever.
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => {}
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, %id, path = %absolute.display(),
|
||||
"could not reclaim deleted media; leaving the row for the next sweep");
|
||||
all_gone = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if !all_gone {
|
||||
continue;
|
||||
}
|
||||
|
||||
if let Err(e) = sqlx::query(
|
||||
"UPDATE upload SET original_path = '', preview_path = NULL,
|
||||
display_path = NULL, thumbnail_path = NULL
|
||||
WHERE id = $1",
|
||||
)
|
||||
.bind(id)
|
||||
.execute(pool)
|
||||
.await
|
||||
{
|
||||
tracing::warn!(error = ?e, %id, "reclaimed the files but could not clear the paths");
|
||||
}
|
||||
}
|
||||
|
||||
if reclaimed > 0 {
|
||||
tracing::info!(
|
||||
"reclaimed {reclaimed} file(s) from soft-deleted uploads (deliberate deletes after \
|
||||
{DELETED_UPLOAD_RETENTION_HOURS}h, compression failures after \
|
||||
{FAILED_ORIGINAL_RETENTION_DAYS}d)"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
async fn cleanup_sessions(pool: &PgPool) {
|
||||
match sqlx::query("DELETE FROM session WHERE expires_at < NOW() - INTERVAL '1 day'")
|
||||
.execute(pool)
|
||||
.await
|
||||
{
|
||||
Ok(r) if r.rows_affected() > 0 => {
|
||||
tracing::info!("cleaned up {} expired session(s)", r.rows_affected());
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(e) => tracing::warn!("session cleanup failed: {e:#}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Reclaim files in `originals/` that no upload row references.
|
||||
///
|
||||
/// The backstop behind [`TempFileGuard`](crate::handlers::upload). The guard covers the
|
||||
/// process that is running; this covers the process that was killed — a SIGKILL, an OOM, or a
|
||||
/// power cut leaves whatever bytes had been written with no `Drop` to reclaim them, and those
|
||||
/// files are then permanently invisible: they have no row, so `cleanup_deleted_media` (which is
|
||||
/// row-driven) can never see them, and they are not counted against any quota while still
|
||||
/// consuming the free disk that `compute_storage_quota` divides among guests. On a single box
|
||||
/// where all three volumes share a filesystem, that ends with Postgres unable to write WAL.
|
||||
///
|
||||
/// Two classes:
|
||||
/// - `*.tmp` — an upload that never got as far as being renamed. Always safe past the window.
|
||||
/// - everything else — a final-named original whose commit never happened.
|
||||
async fn sweep_orphan_originals(pool: &PgPool, media_path: &std::path::Path) {
|
||||
let originals = media_path.join("originals");
|
||||
let cutoff = Duration::from_secs(ORPHAN_UPLOAD_RETENTION_HOURS * 3600);
|
||||
|
||||
// originals/{event_slug}/{uuid}.{ext} — one level of per-event directories.
|
||||
let mut event_dirs = match tokio::fs::read_dir(&originals).await {
|
||||
Ok(rd) => rd,
|
||||
// Nothing uploaded yet; the directory is created lazily by the upload handler.
|
||||
Err(_) => return,
|
||||
};
|
||||
|
||||
let mut candidates: Vec<(String, std::path::PathBuf)> = Vec::new();
|
||||
let mut temps_removed = 0u32;
|
||||
|
||||
while let Ok(Some(event_dir)) = event_dirs.next_entry().await {
|
||||
if !event_dir
|
||||
.file_type()
|
||||
.await
|
||||
.map(|t| t.is_dir())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let slug = event_dir.file_name().to_string_lossy().to_string();
|
||||
let Ok(mut files) = tokio::fs::read_dir(event_dir.path()).await else {
|
||||
continue;
|
||||
};
|
||||
while let Ok(Some(entry)) = files.next_entry().await {
|
||||
let Ok(meta) = entry.metadata().await else {
|
||||
continue;
|
||||
};
|
||||
if !meta.is_file() {
|
||||
continue;
|
||||
}
|
||||
// Too young to judge: an upload committing RIGHT NOW is indistinguishable from an
|
||||
// orphan, because the rename precedes the commit.
|
||||
let recent = meta
|
||||
.modified()
|
||||
.ok()
|
||||
.and_then(|m| m.elapsed().ok())
|
||||
.is_none_or(|age| age < cutoff);
|
||||
if recent {
|
||||
continue;
|
||||
}
|
||||
|
||||
let name = entry.file_name().to_string_lossy().to_string();
|
||||
if name.ends_with(".tmp") {
|
||||
// A `.tmp` never has a row by construction — no DB check needed.
|
||||
if tokio::fs::remove_file(entry.path()).await.is_ok() {
|
||||
temps_removed += 1;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
candidates.push((format!("originals/{slug}/{name}"), entry.path()));
|
||||
}
|
||||
}
|
||||
|
||||
if temps_removed > 0 {
|
||||
tracing::warn!(
|
||||
"reclaimed {temps_removed} abandoned upload temp file(s) older than \
|
||||
{ORPHAN_UPLOAD_RETENTION_HOURS}h"
|
||||
);
|
||||
}
|
||||
if candidates.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
// One query per batch, not one per file: a backlog of thousands of orphans must not turn
|
||||
// into thousands of round trips on an hourly timer.
|
||||
let mut orphans_removed = 0u32;
|
||||
for chunk in candidates.chunks(500) {
|
||||
let paths: Vec<String> = chunk.iter().map(|(rel, _)| rel.clone()).collect();
|
||||
// NO `deleted_at IS NULL` FILTER HERE. A soft-deleted row still points at its file
|
||||
// during its retention window, and reclaiming that file is `cleanup_deleted_media`'s
|
||||
// job — filtering here would race the two sweeps and destroy the exact files the
|
||||
// recovery window exists to preserve.
|
||||
let unreferenced: Result<Vec<(String,)>, _> = sqlx::query_as(
|
||||
"SELECT p FROM unnest($1::text[]) AS p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM upload u WHERE u.original_path = p)",
|
||||
)
|
||||
.bind(&paths)
|
||||
.fetch_all(pool)
|
||||
.await;
|
||||
let unreferenced = match unreferenced {
|
||||
Ok(rows) => rows,
|
||||
Err(e) => {
|
||||
tracing::warn!(error = ?e, "orphan-original sweep query failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
for (rel,) in unreferenced {
|
||||
if let Some((_, abs)) = chunk.iter().find(|(r, _)| *r == rel)
|
||||
&& tokio::fs::remove_file(abs).await.is_ok()
|
||||
{
|
||||
orphans_removed += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if orphans_removed > 0 {
|
||||
tracing::warn!(
|
||||
"reclaimed {orphans_removed} original(s) with no upload row, older than \
|
||||
{ORPHAN_UPLOAD_RETENTION_HOURS}h"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,3 +1,9 @@
|
||||
pub mod compression;
|
||||
pub mod config;
|
||||
pub mod disk;
|
||||
pub mod export;
|
||||
pub mod imaging;
|
||||
pub mod maintenance;
|
||||
pub mod rate_limiter;
|
||||
pub mod sse_tickets;
|
||||
pub mod video;
|
||||
|
||||
@@ -17,14 +17,20 @@ impl RateLimiter {
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns `true` if the request is allowed, `false` if rate-limited.
|
||||
pub fn check(&self, key: impl Into<String>, max: usize, window: Duration) -> bool {
|
||||
self.check_with_retry(key, max, window).is_ok()
|
||||
}
|
||||
|
||||
/// Returns `Ok(())` if allowed, `Err(retry_after_secs)` if rate-limited.
|
||||
/// `retry_after_secs` is how long until the oldest slot in the window expires.
|
||||
pub fn check_with_retry(&self, key: impl Into<String>, max: usize, window: Duration) -> Result<(), u64> {
|
||||
///
|
||||
/// This is deliberately the ONLY entry point. There used to be a `check()` wrapper
|
||||
/// returning a plain bool, and 7 of the 8 call sites used it and then hard-coded
|
||||
/// `None` for the response's `Retry-After` — so a throttled client was told to back
|
||||
/// off but never for how long. Forcing every caller through the `Result` makes the
|
||||
/// retry delay impossible to discard by accident.
|
||||
pub fn check_with_retry(
|
||||
&self,
|
||||
key: impl Into<String>,
|
||||
max: usize,
|
||||
window: Duration,
|
||||
) -> Result<(), u64> {
|
||||
let now = Instant::now();
|
||||
let key = key.into();
|
||||
let mut map = self.windows.lock().unwrap();
|
||||
@@ -41,15 +47,262 @@ impl RateLimiter {
|
||||
Err(remaining.as_secs().max(1))
|
||||
}
|
||||
}
|
||||
|
||||
/// Wipe every tracked window. Used by the test-mode truncate route so a previous
|
||||
/// test's accumulated counters don't bleed into the next test's rate-limit checks.
|
||||
pub fn clear(&self) {
|
||||
self.windows.lock().unwrap().clear();
|
||||
}
|
||||
|
||||
/// Drop keys whose windows are empty after expiring old timestamps. Called from a
|
||||
/// background task (see [`crate::services::maintenance`]) so that long-lived
|
||||
/// processes don't accumulate one HashMap entry per IP that ever connected.
|
||||
///
|
||||
/// Uses a conservative 24h ceiling — anything older than that is gone regardless
|
||||
/// of which endpoint's window it was tracked under (the longest window today is
|
||||
/// 24h for export downloads). If we ever add longer windows, raise this constant.
|
||||
pub fn prune(&self) {
|
||||
let now = Instant::now();
|
||||
let ceiling = Duration::from_secs(24 * 60 * 60);
|
||||
let mut map = self.windows.lock().unwrap();
|
||||
let before = map.len();
|
||||
map.retain(|_, ts| {
|
||||
ts.retain(|&t| now.duration_since(t) < ceiling);
|
||||
!ts.is_empty()
|
||||
});
|
||||
let dropped = before.saturating_sub(map.len());
|
||||
if dropped > 0 {
|
||||
tracing::debug!("rate limiter pruned {dropped} idle keys");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the client IP from X-Forwarded-For (Caddy sets this) or fall back
|
||||
/// to a provided socket address string.
|
||||
/// Extract the client IP from X-Forwarded-For or fall back to a provided socket
|
||||
/// address string.
|
||||
///
|
||||
/// We take the **right-most** entry, not the left-most. Caddy is the sole ingress
|
||||
/// and the app port is only `expose`d (never published), so the last hop Caddy
|
||||
/// appends is the real client. A client can prepend arbitrary spoofed values to
|
||||
/// the left of XFF to dodge throttles — those are ignored here. This assumes
|
||||
/// exactly one trusted proxy (Caddy); revisit if that changes.
|
||||
///
|
||||
/// Pass the peer address as `fallback`, never a constant. Every caller used to pass
|
||||
/// the literal `"unknown"`, so any request that arrived without XFF — i.e. anything
|
||||
/// reaching the app directly rather than through Caddy — shared ONE bucket with every
|
||||
/// other such request, turning the limiter into a self-inflicted global throttle.
|
||||
pub fn client_ip(headers: &axum::http::HeaderMap, fallback: &str) -> String {
|
||||
headers
|
||||
.get("x-forwarded-for")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.and_then(|s| s.split(',').next())
|
||||
.and_then(|s| s.rsplit(',').next())
|
||||
.map(|s| s.trim().to_owned())
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or_else(|| fallback.to_owned())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use axum::http::HeaderMap;
|
||||
|
||||
const MIN: Duration = Duration::from_secs(60);
|
||||
|
||||
#[test]
|
||||
fn allows_up_to_max_then_blocks() {
|
||||
let rl = RateLimiter::new();
|
||||
assert!(rl.check_with_retry("k", 3, MIN).is_ok());
|
||||
assert!(rl.check_with_retry("k", 3, MIN).is_ok());
|
||||
assert!(rl.check_with_retry("k", 3, MIN).is_ok());
|
||||
assert!(
|
||||
rl.check_with_retry("k", 3, MIN).is_err(),
|
||||
"the 4th request must be blocked"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_are_independent() {
|
||||
let rl = RateLimiter::new();
|
||||
assert!(rl.check_with_retry("a", 1, MIN).is_ok());
|
||||
assert!(rl.check_with_retry("a", 1, MIN).is_err());
|
||||
assert!(
|
||||
rl.check_with_retry("b", 1, MIN).is_ok(),
|
||||
"a different key has its own window"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn window_slides_and_allows_again_after_expiry() {
|
||||
let rl = RateLimiter::new();
|
||||
let w = Duration::from_millis(40);
|
||||
assert!(rl.check_with_retry("k", 1, w).is_ok());
|
||||
assert!(rl.check_with_retry("k", 1, w).is_err());
|
||||
std::thread::sleep(Duration::from_millis(55));
|
||||
assert!(
|
||||
rl.check_with_retry("k", 1, w).is_ok(),
|
||||
"the slot should expire once the window passes"
|
||||
);
|
||||
}
|
||||
|
||||
/// `retry_after` is not a "some number in range" — it is the time until the oldest slot
|
||||
/// in the window frees up, and it is surfaced to clients as the backoff they sleep for
|
||||
/// (see `upload-queue.ts`). Asserting only `(1..=60)` spans the entire reachable domain
|
||||
/// of a 60s window, so a hardcoded `Err(1)` would satisfy it while telling every client
|
||||
/// to hammer the server a second later. Pin the actual value.
|
||||
#[test]
|
||||
fn retry_after_is_the_remaining_window() {
|
||||
let rl = RateLimiter::new();
|
||||
|
||||
// The slot was consumed just now, so essentially the whole window remains.
|
||||
// `as_secs()` truncates the sub-second remainder, so a 30s window reports 29.
|
||||
let w30 = Duration::from_secs(30);
|
||||
assert!(rl.check_with_retry("a", 1, w30).is_ok());
|
||||
let a = rl.check_with_retry("a", 1, w30).unwrap_err();
|
||||
assert_eq!(a, 29, "retry_after must be the remaining window, got {a}");
|
||||
|
||||
// A different window must yield a different retry_after: no single constant can
|
||||
// satisfy both this and the assertion above.
|
||||
let w10 = Duration::from_secs(10);
|
||||
assert!(rl.check_with_retry("b", 1, w10).is_ok());
|
||||
let b = rl.check_with_retry("b", 1, w10).unwrap_err();
|
||||
assert_eq!(b, 9, "retry_after must scale with the window, got {b}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retry_after_counts_down_as_the_window_elapses() {
|
||||
let rl = RateLimiter::new();
|
||||
let w = Duration::from_secs(30);
|
||||
assert!(rl.check_with_retry("k", 1, w).is_ok());
|
||||
let first = rl.check_with_retry("k", 1, w).unwrap_err();
|
||||
|
||||
std::thread::sleep(Duration::from_millis(1200));
|
||||
let second = rl.check_with_retry("k", 1, w).unwrap_err();
|
||||
|
||||
// A client that waits 1.2s must be told to wait ~1.2s less — otherwise the advertised
|
||||
// backoff is a constant, not a deadline.
|
||||
let shaved = first - second;
|
||||
assert!(
|
||||
(1..=2).contains(&shaved),
|
||||
"1.2s of waiting must shorten the advertised backoff by ~1s (got {first} then {second})"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retry_after_floors_at_one_second() {
|
||||
let rl = RateLimiter::new();
|
||||
let w = Duration::from_millis(800);
|
||||
assert!(rl.check_with_retry("k", 1, w).is_ok());
|
||||
let retry = rl.check_with_retry("k", 1, w).unwrap_err();
|
||||
// The sub-second remainder truncates to 0; clients must never be told "retry in 0s"
|
||||
// (that's a busy-loop). The `.max(1)` floor is what prevents it.
|
||||
assert_eq!(
|
||||
retry, 1,
|
||||
"a sub-second remainder must floor to 1, got {retry}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clear_resets_every_window() {
|
||||
let rl = RateLimiter::new();
|
||||
assert!(rl.check_with_retry("k", 1, MIN).is_ok());
|
||||
assert!(rl.check_with_retry("k", 1, MIN).is_err());
|
||||
rl.clear();
|
||||
assert!(
|
||||
rl.check_with_retry("k", 1, MIN).is_ok(),
|
||||
"clear() must free the window"
|
||||
);
|
||||
}
|
||||
|
||||
/// `prune()` is a memory-leak guard: without it a long-lived process keeps one HashMap
|
||||
/// entry per IP that ever connected. Nothing in the public API observes the map size, so
|
||||
/// the only way to catch a no-op body (`fn prune(&self) {}`) is to look at the map — the
|
||||
/// tests module can see the private field.
|
||||
#[test]
|
||||
fn prune_drops_keys_whose_windows_have_fully_expired() {
|
||||
let rl = RateLimiter::new();
|
||||
|
||||
// A key whose only timestamp is older than the 24h ceiling. We can't sleep for a day,
|
||||
// so backdate the Instant directly.
|
||||
let ancient = Instant::now()
|
||||
.checked_sub(Duration::from_secs(25 * 60 * 60))
|
||||
.expect("backdating an Instant by 25h");
|
||||
rl.windows
|
||||
.lock()
|
||||
.unwrap()
|
||||
.insert("stale".to_string(), vec![ancient]);
|
||||
|
||||
// ...alongside a key that is still inside its window.
|
||||
assert!(rl.check_with_retry("live", 5, MIN).is_ok());
|
||||
assert_eq!(rl.windows.lock().unwrap().len(), 2);
|
||||
|
||||
rl.prune();
|
||||
|
||||
let map = rl.windows.lock().unwrap();
|
||||
assert!(
|
||||
!map.contains_key("stale"),
|
||||
"prune() must drop keys whose timestamps have all expired"
|
||||
);
|
||||
assert!(
|
||||
map.contains_key("live"),
|
||||
"prune() must keep keys that still have live timestamps"
|
||||
);
|
||||
assert_eq!(map.len(), 1, "exactly one key should survive the prune");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prune_does_not_reset_a_live_window() {
|
||||
// The counterpart to the test above: pruning must reclaim memory, never quota. If
|
||||
// prune() dropped live keys, every background sweep would hand attackers a fresh
|
||||
// budget.
|
||||
let rl = RateLimiter::new();
|
||||
assert!(rl.check_with_retry("k", 1, MIN).is_ok());
|
||||
assert!(rl.check_with_retry("k", 1, MIN).is_err());
|
||||
|
||||
rl.prune();
|
||||
|
||||
assert!(
|
||||
rl.check_with_retry("k", 1, MIN).is_err(),
|
||||
"prune() must not clear a window that is still active"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn client_ip_takes_rightmost_forwarded_for_entry() {
|
||||
// The right-most entry is the hop our trusted proxy (Caddy) appended.
|
||||
let mut h = HeaderMap::new();
|
||||
h.insert("x-forwarded-for", "10.0.0.1, 203.0.113.7".parse().unwrap());
|
||||
assert_eq!(client_ip(&h, "fallback"), "203.0.113.7");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn client_ip_ignores_spoofed_leftmost_entry() {
|
||||
// A client prepending a fake IP to dodge throttles must not win.
|
||||
let mut h = HeaderMap::new();
|
||||
h.insert(
|
||||
"x-forwarded-for",
|
||||
"1.2.3.4, 9.9.9.9, 203.0.113.7".parse().unwrap(),
|
||||
);
|
||||
assert_eq!(client_ip(&h, "fallback"), "203.0.113.7");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn client_ip_trims_surrounding_whitespace() {
|
||||
let mut h = HeaderMap::new();
|
||||
h.insert("x-forwarded-for", " 198.51.100.5 ".parse().unwrap());
|
||||
assert_eq!(client_ip(&h, "fb"), "198.51.100.5");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn client_ip_falls_back_when_header_absent() {
|
||||
assert_eq!(client_ip(&HeaderMap::new(), "127.0.0.1"), "127.0.0.1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn client_ip_falls_back_on_trailing_comma_empty_entry() {
|
||||
// A trailing comma leaves an empty right-most segment after trimming; the
|
||||
// `.filter(!is_empty)` must reject it and fall through to the fallback
|
||||
// rather than returning "" (which would collapse callers into one bucket).
|
||||
let mut h = HeaderMap::new();
|
||||
h.insert("x-forwarded-for", "203.0.113.7, ".parse().unwrap());
|
||||
assert_eq!(client_ip(&h, "127.0.0.1"), "127.0.0.1");
|
||||
}
|
||||
}
|
||||
|
||||
273
backend/src/services/sse_tickets.rs
Normal file
273
backend/src/services/sse_tickets.rs
Normal file
@@ -0,0 +1,273 @@
|
||||
use std::collections::HashMap;
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use rand::Rng;
|
||||
|
||||
/// Short-lived single-use tickets that let `EventSource` clients open the SSE
|
||||
/// stream without putting the JWT in the URL (where it would leak via access
|
||||
/// logs / referer / browser history).
|
||||
///
|
||||
/// Flow: client `POST /api/v1/stream/ticket` with `Authorization: Bearer <jwt>`,
|
||||
/// server returns an opaque ticket, client passes it as `?ticket=...` on the
|
||||
/// stream open. Tickets are consumed on use and expire after `TTL`.
|
||||
const TTL: Duration = Duration::from_secs(30);
|
||||
|
||||
/// Ceiling on outstanding tickets across the whole process.
|
||||
///
|
||||
/// Not really about the bytes (~120 each) — about `issue` having had no bound of any kind.
|
||||
/// Sized well above a real event: ~1000 concurrent clients each holding one live 30 s ticket.
|
||||
const MAX_TICKETS: usize = 4096;
|
||||
|
||||
/// Live tickets one session may hold. Above 1 because two tabs sharing a token open their
|
||||
/// EventSources concurrently; 4 absorbs that without letting a reconnect loop accumulate.
|
||||
const MAX_TICKETS_PER_SESSION: usize = 4;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct SseTicketStore {
|
||||
inner: Arc<Mutex<HashMap<String, Entry>>>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct Entry {
|
||||
token_hash: String,
|
||||
issued_at: Instant,
|
||||
}
|
||||
|
||||
impl SseTicketStore {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(Mutex::new(HashMap::new())),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop every outstanding ticket. Used by the e2e TRUNCATE endpoint: tickets are bound to a
|
||||
/// session token hash, and TRUNCATE deletes the sessions out from under them, so anything left
|
||||
/// here is a dangling reference to a user that no longer exists.
|
||||
pub fn clear(&self) {
|
||||
self.inner.lock().unwrap().clear();
|
||||
}
|
||||
|
||||
/// Mint a new ticket bound to the caller's session (identified by token hash).
|
||||
///
|
||||
/// `None` when the store is at capacity — the caller should answer 503, not evict.
|
||||
///
|
||||
/// Three bounds, because `issue` had none: no size cap, no per-caller cap, and no rate
|
||||
/// limit on the endpoint, while `prune` ran only hourly against a 30-second TTL. So any
|
||||
/// authenticated session could loop the endpoint and grow the map for an hour.
|
||||
pub fn issue(&self, token_hash: String) -> Option<String> {
|
||||
let ticket = random_ticket();
|
||||
let mut map = self.inner.lock().unwrap();
|
||||
|
||||
// Prune on issue rather than only hourly. This alone changes the bound from "tickets
|
||||
// minted since the last maintenance tick" to "tickets live at once", which is what the
|
||||
// 30 s TTL was always meant to express.
|
||||
map.retain(|_, e| e.issued_at.elapsed() <= TTL);
|
||||
|
||||
// Cap the caller's own outstanding tickets, evicting their oldest. NOT one-per-session:
|
||||
// two tabs sharing a token open their EventSources concurrently, and having tab B
|
||||
// invalidate tab A's unconsumed ticket looks exactly like a flaky SSE connection.
|
||||
let mut mine: Vec<(String, Instant)> = map
|
||||
.iter()
|
||||
.filter(|(_, e)| e.token_hash == token_hash)
|
||||
.map(|(k, e)| (k.clone(), e.issued_at))
|
||||
.collect();
|
||||
if mine.len() >= MAX_TICKETS_PER_SESSION {
|
||||
mine.sort_by_key(|(_, issued)| *issued);
|
||||
for (key, _) in mine.iter().take(mine.len() - MAX_TICKETS_PER_SESSION + 1) {
|
||||
map.remove(key);
|
||||
}
|
||||
}
|
||||
|
||||
// At capacity, REFUSE — never evict a stranger's ticket. Evicting would let one
|
||||
// misbehaving client deny SSE to the whole venue, which is worse than failing the
|
||||
// request that hit the ceiling.
|
||||
if map.len() >= MAX_TICKETS {
|
||||
tracing::warn!(
|
||||
outstanding = map.len(),
|
||||
"SSE ticket store at capacity; refusing to mint"
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
map.insert(
|
||||
ticket.clone(),
|
||||
Entry {
|
||||
token_hash,
|
||||
issued_at: Instant::now(),
|
||||
},
|
||||
);
|
||||
Some(ticket)
|
||||
}
|
||||
|
||||
/// Consume a ticket. Returns `Some(token_hash)` if the ticket exists and is
|
||||
/// not expired. Single-use: the ticket is removed regardless of whether it
|
||||
/// was still fresh, so a replay can't slip through after expiry.
|
||||
pub fn consume(&self, ticket: &str) -> Option<String> {
|
||||
let mut map = self.inner.lock().unwrap();
|
||||
let entry = map.remove(ticket)?;
|
||||
if entry.issued_at.elapsed() > TTL {
|
||||
return None;
|
||||
}
|
||||
Some(entry.token_hash)
|
||||
}
|
||||
|
||||
/// Drop expired entries — called from the background maintenance task so a
|
||||
/// long-running process doesn't accumulate stale tickets.
|
||||
pub fn prune(&self) {
|
||||
let mut map = self.inner.lock().unwrap();
|
||||
map.retain(|_, e| e.issued_at.elapsed() <= TTL);
|
||||
}
|
||||
}
|
||||
|
||||
fn random_ticket() -> String {
|
||||
// 192 bits of randomness, base32-ish hex. Plenty of entropy and URL-safe.
|
||||
let mut rng = rand::rng();
|
||||
let mut bytes = [0u8; 24];
|
||||
rng.fill(&mut bytes);
|
||||
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// `issue` now returns `Option`; in every test below the store is far from capacity, so an
|
||||
/// `expect` here documents that refusing is exceptional rather than routine.
|
||||
fn issue(store: &SseTicketStore, hash: &str) -> String {
|
||||
store.issue(hash.into()).expect("store has capacity")
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn issue_then_consume_returns_the_hash_exactly_once() {
|
||||
let store = SseTicketStore::new();
|
||||
let ticket = issue(&store, "hash-1");
|
||||
assert_eq!(store.consume(&ticket).as_deref(), Some("hash-1"));
|
||||
// Single-use: a replay of the same ticket is rejected.
|
||||
assert_eq!(
|
||||
store.consume(&ticket),
|
||||
None,
|
||||
"a consumed ticket must not be reusable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_ticket_consumes_to_none() {
|
||||
let store = SseTicketStore::new();
|
||||
assert_eq!(store.consume("never-issued"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn issued_tickets_are_unique_and_hex() {
|
||||
let store = SseTicketStore::new();
|
||||
let a = issue(&store, "h");
|
||||
let b = issue(&store, "h");
|
||||
assert_ne!(a, b, "each ticket must be unique");
|
||||
assert_eq!(a.len(), 48, "24 random bytes → 48 hex chars");
|
||||
assert!(a.chars().all(|c| c.is_ascii_hexdigit()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fresh_ticket_survives_prune() {
|
||||
let store = SseTicketStore::new();
|
||||
let ticket = issue(&store, "h");
|
||||
store.prune(); // not expired → kept
|
||||
assert_eq!(store.consume(&ticket).as_deref(), Some("h"));
|
||||
}
|
||||
|
||||
/// Build an entry that is already past the TTL.
|
||||
fn insert_stale(store: &SseTicketStore, key: &str, token_hash: &str) {
|
||||
store.inner.lock().unwrap().insert(
|
||||
key.to_string(),
|
||||
Entry {
|
||||
token_hash: token_hash.into(),
|
||||
issued_at: Instant::now()
|
||||
.checked_sub(TTL + Duration::from_secs(1))
|
||||
.expect("host uptime should exceed the ticket TTL"),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn expired_ticket_consumes_to_none() {
|
||||
let store = SseTicketStore::new();
|
||||
insert_stale(&store, "stale-ticket", "h");
|
||||
assert_eq!(
|
||||
store.consume("stale-ticket"),
|
||||
None,
|
||||
"an expired ticket must not authenticate"
|
||||
);
|
||||
}
|
||||
|
||||
/// The TTL is 30 s but `prune` only ran hourly, so the map was really bounded by "tickets
|
||||
/// minted in the last hour" — which is unbounded for a client in a loop.
|
||||
#[test]
|
||||
fn issuing_prunes_expired_entries() {
|
||||
let store = SseTicketStore::new();
|
||||
insert_stale(&store, "stale-a", "someone-else");
|
||||
insert_stale(&store, "stale-b", "someone-else");
|
||||
issue(&store, "h");
|
||||
assert_eq!(
|
||||
store.inner.lock().unwrap().len(),
|
||||
1,
|
||||
"issue must reclaim expired slots, not merely add to them"
|
||||
);
|
||||
}
|
||||
|
||||
/// Two tabs sharing a token is normal, so the per-session cap must be above 1 — but a
|
||||
/// reconnect loop must not accumulate. The caller's OWN oldest is what gets evicted.
|
||||
#[test]
|
||||
fn a_session_is_capped_and_evicts_only_its_own_oldest() {
|
||||
let store = SseTicketStore::new();
|
||||
let stranger = issue(&store, "other-session");
|
||||
|
||||
let mut mine: Vec<String> = Vec::new();
|
||||
for _ in 0..MAX_TICKETS_PER_SESSION + 2 {
|
||||
mine.push(issue(&store, "mine"));
|
||||
}
|
||||
|
||||
let live = mine
|
||||
.iter()
|
||||
.filter(|t| store.inner.lock().unwrap().contains_key(*t))
|
||||
.count();
|
||||
assert_eq!(live, MAX_TICKETS_PER_SESSION, "one session, bounded");
|
||||
assert!(
|
||||
store.inner.lock().unwrap().contains_key(&mine[mine.len() - 1]),
|
||||
"the newest ticket is the one the caller is about to use"
|
||||
);
|
||||
assert_eq!(
|
||||
store.consume(&stranger).as_deref(),
|
||||
Some("other-session"),
|
||||
"another session's ticket must survive — evicting it would let one client deny \
|
||||
SSE to the venue"
|
||||
);
|
||||
}
|
||||
|
||||
/// At capacity the store REFUSES rather than evicting a stranger. Refusing fails the one
|
||||
/// request that hit the ceiling; evicting would break an unrelated client's live stream.
|
||||
#[test]
|
||||
fn at_capacity_the_store_refuses_instead_of_evicting() {
|
||||
let store = SseTicketStore::new();
|
||||
{
|
||||
let mut map = store.inner.lock().unwrap();
|
||||
for i in 0..MAX_TICKETS {
|
||||
map.insert(
|
||||
format!("filler-{i}"),
|
||||
Entry {
|
||||
token_hash: format!("session-{i}"),
|
||||
issued_at: Instant::now(),
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
assert_eq!(
|
||||
store.issue("newcomer".into()),
|
||||
None,
|
||||
"a full store must refuse, so the caller can answer 503"
|
||||
);
|
||||
assert!(
|
||||
store.inner.lock().unwrap().contains_key("filler-0"),
|
||||
"no existing ticket may be sacrificed to make room"
|
||||
);
|
||||
}
|
||||
}
|
||||
233
backend/src/services/video.rs
Normal file
233
backend/src/services/video.rs
Normal file
@@ -0,0 +1,233 @@
|
||||
//! Poster-frame extraction, shared by the compression worker and the HTML export.
|
||||
//!
|
||||
//! Both used to spawn `ffmpeg` themselves with the same broken invocation:
|
||||
//!
|
||||
//! ```text
|
||||
//! ffmpeg -i <src> -vframes 1 -ss 00:00:01 -vf scale=… -y <out>
|
||||
//! ```
|
||||
//!
|
||||
//! `-ss` AFTER `-i` is an output-side seek. Against a clip of a second or less ffmpeg exits **0 and
|
||||
//! writes nothing** — and both call sites gated on the exit status, so neither noticed. The worker
|
||||
//! then wrote `thumbnail_path` for a file that was never created (404 in the live feed) and the
|
||||
//! export listed the entry in `data.json` while the ZIP writer skipped it (a broken image tile in
|
||||
//! the keepsake). Every server-side signal stayed green. Phones produce such clips constantly:
|
||||
//! mis-taps, Live Photos, boomerangs.
|
||||
//!
|
||||
//! This module exists for the same reason `imaging.rs` does — that one was created when compression
|
||||
//! and export duplicated decode logic, and it paid off immediately when the `max_alloc` fix landed
|
||||
//! in both workers at once. Same duplication, same fix.
|
||||
|
||||
use std::path::Path;
|
||||
use std::time::Duration;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
|
||||
/// A malformed video can hang `ffmpeg` indefinitely. In the compression worker that never releases
|
||||
/// the semaphore permit and the pool eventually deadlocks; in the export worker it strands the job
|
||||
/// at `running` so the keepsake never completes. `export.rs` had NO timeout at all before this
|
||||
/// module — sharing the spawn fixes that too.
|
||||
const FFMPEG_TIMEOUT: Duration = Duration::from_secs(120);
|
||||
|
||||
/// Seek positions to try, in order.
|
||||
///
|
||||
/// One second first: the opening frame of a real video is often black, a fade-in, or motion-blurred
|
||||
/// as the camera settles, so it makes a poor poster. Zero second as the fallback, which is what
|
||||
/// makes short clips work — and it is genuinely required, not defensive. Moving `-ss` before `-i`
|
||||
/// (an input-side seek) is necessary but NOT sufficient: seeking to 1 s in a 1.000 s clip is still
|
||||
/// past the last frame, and ffmpeg still exits 0 having written nothing. Verified against the real
|
||||
/// production image.
|
||||
const SEEK_POSITIONS: &[&str] = &["00:00:01", "0"];
|
||||
|
||||
/// Extract one poster frame from `src` into `dest`, scaled to `width` px wide.
|
||||
///
|
||||
/// `Ok(false)` means the video yielded no frame — a normal outcome for a very short or unusual
|
||||
/// clip, NOT an error. Callers must degrade (no poster) rather than fail the upload: treating this
|
||||
/// as an error would soft-delete every sub-second video, turning a cosmetic defect into data loss.
|
||||
///
|
||||
/// `Err` is reserved for something genuinely wrong — a hang we had to kill, or a failure to spawn.
|
||||
pub async fn extract_poster_frame(src: &Path, dest: &Path, width: u32) -> Result<bool> {
|
||||
for seek in SEEK_POSITIONS {
|
||||
// A stale file from a previous attempt would be indistinguishable from a fresh success.
|
||||
let _ = tokio::fs::remove_file(dest).await;
|
||||
|
||||
run_ffmpeg(src, dest, width, seek).await?;
|
||||
|
||||
// THE CHECK BOTH CALL SITES WERE MISSING: ask the filesystem, not the exit status.
|
||||
// Non-empty, because a zero-byte file is not a poster either.
|
||||
if tokio::fs::metadata(dest)
|
||||
.await
|
||||
.map(|m| m.is_file() && m.len() > 0)
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return Ok(true);
|
||||
}
|
||||
}
|
||||
|
||||
// Leave nothing behind for a caller to mistake for a result.
|
||||
let _ = tokio::fs::remove_file(dest).await;
|
||||
Ok(false)
|
||||
}
|
||||
|
||||
/// Run one ffmpeg attempt. A non-zero exit is NOT an error here — the artifact check above is the
|
||||
/// authority, and a corrupt input that fails at 1 s may still yield a frame at 0.
|
||||
async fn run_ffmpeg(src: &Path, dest: &Path, width: u32, seek: &str) -> Result<()> {
|
||||
let child = tokio::process::Command::new("ffmpeg")
|
||||
.args([
|
||||
// BEFORE -i: an input-side seek. See SEEK_POSITIONS.
|
||||
"-ss",
|
||||
seek,
|
||||
"-i",
|
||||
src.to_str().unwrap_or_default(),
|
||||
"-vframes",
|
||||
"1",
|
||||
"-vf",
|
||||
&format!("scale={width}:-1"),
|
||||
"-y",
|
||||
dest.to_str().unwrap_or_default(),
|
||||
])
|
||||
// ffmpeg writes the poster to `dest` itself; nothing here ever reads stdout, so
|
||||
// giving it a pipe only created something that could fill.
|
||||
.stdout(std::process::Stdio::null())
|
||||
// stderr IS piped — it is the only diagnostic when a clip yields no frame — but it
|
||||
// must be DRAINED, which is the whole point of `wait_with_output` below.
|
||||
.stderr(std::process::Stdio::piped())
|
||||
.kill_on_drop(true)
|
||||
.spawn()
|
||||
.context("failed to spawn ffmpeg")?;
|
||||
|
||||
// `wait_with_output`, NOT `wait`. ffmpeg is verbose on stderr (banner, stream info,
|
||||
// per-frame progress) and `wait()` reads neither pipe — so once the ~64 KiB pipe buffer
|
||||
// filled, ffmpeg blocked writing, `wait()` never returned, and the call burned the full
|
||||
// timeout. That is not merely slow: the timeout is an `Err`, so after 2 seek positions x
|
||||
// 3 compression attempts the caller soft-deletes a perfectly playable video for a
|
||||
// poster-frame failure. `wait_with_output` polls the pipe and the exit status together.
|
||||
//
|
||||
// It also CONSUMES the child, so the explicit `child.kill()` that used to sit on the
|
||||
// timeout arm cannot exist here — and is not needed: `kill_on_drop(true)` is set above,
|
||||
// and dropping the future on timeout drops the child with it.
|
||||
let out = match tokio::time::timeout(FFMPEG_TIMEOUT, child.wait_with_output()).await {
|
||||
Ok(res) => res.context("ffmpeg wait failed")?,
|
||||
Err(_) => anyhow::bail!("ffmpeg timed out after {}s", FFMPEG_TIMEOUT.as_secs()),
|
||||
};
|
||||
|
||||
// A non-zero exit is not an error (see the doc comment) — the artifact check in
|
||||
// `extract_poster_frame` is the authority. Log the tail so a systematically failing
|
||||
// format is diagnosable without turning it into data loss.
|
||||
if !out.status.success() {
|
||||
tracing::debug!(
|
||||
seek,
|
||||
status = ?out.status,
|
||||
stderr = %tail_lines(&out.stderr, 10),
|
||||
"ffmpeg exited non-zero; the artifact check decides"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Last `n` lines of a child's stderr, lossily decoded.
|
||||
///
|
||||
/// Bounded on purpose: ffmpeg's stderr is unbounded, and the reason we now drain it is that
|
||||
/// unbounded output used to be a hazard. Emitting all of it into a log line — into container
|
||||
/// logs that are themselves size-capped — would just move the problem.
|
||||
fn tail_lines(bytes: &[u8], n: usize) -> String {
|
||||
let text = String::from_utf8_lossy(bytes);
|
||||
let lines: Vec<&str> = text.lines().filter(|l| !l.trim().is_empty()).collect();
|
||||
lines[lines.len().saturating_sub(n)..].join(" | ")
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The order is the whole fix. `-ss` must precede `-i`, and 0 must be tried after 1 s.
|
||||
#[test]
|
||||
fn the_fallback_seek_exists_and_comes_last() {
|
||||
assert_eq!(
|
||||
SEEK_POSITIONS,
|
||||
&["00:00:01", "0"],
|
||||
"1s first for a better poster, 0 as the fallback that makes short clips work"
|
||||
);
|
||||
}
|
||||
|
||||
/// A missing input yields no frame rather than an error: the caller must degrade to "no
|
||||
/// poster", never fail the upload. `Err` is reserved for a hang or a spawn failure.
|
||||
#[tokio::test]
|
||||
async fn a_missing_source_yields_no_frame_rather_than_an_error() {
|
||||
let dir = std::env::temp_dir().join(format!("es-video-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let dest = dir.join("out.jpg");
|
||||
|
||||
let got = extract_poster_frame(Path::new("/nonexistent/clip.mp4"), &dest, 400).await;
|
||||
|
||||
match got {
|
||||
Ok(false) => {}
|
||||
other => panic!("expected Ok(false) for a missing input, got {other:?}"),
|
||||
}
|
||||
assert!(
|
||||
!dest.exists(),
|
||||
"a failed extraction must leave nothing a caller could mistake for a poster"
|
||||
);
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_stderr_tail_is_bounded_and_survives_invalid_utf8() {
|
||||
let noisy: Vec<u8> = (0..500)
|
||||
.map(|i| format!("line {i}\n"))
|
||||
.collect::<String>()
|
||||
.into_bytes();
|
||||
let got = tail_lines(&noisy, 3);
|
||||
assert_eq!(got, "line 497 | line 498 | line 499");
|
||||
|
||||
// ffmpeg emits filenames verbatim, so its stderr is not guaranteed to be UTF-8.
|
||||
assert_eq!(tail_lines(&[b'o', b'k', 0xff], 5), "ok\u{fffd}");
|
||||
assert_eq!(tail_lines(b"", 5), "");
|
||||
}
|
||||
|
||||
/// A real extraction must finish in a small fraction of `FFMPEG_TIMEOUT`.
|
||||
///
|
||||
/// Wall-clock is the ONLY observable of the bug this guards: piping stderr and then
|
||||
/// calling `wait()` (which drains nothing) blocks ffmpeg on a full pipe buffer until the
|
||||
/// timeout fires, and the timeout is an `Err`, so the upload is soft-deleted. The
|
||||
/// assertion is deliberately on elapsed time, not on the exit status.
|
||||
///
|
||||
/// Honest limitation: our fixture is quiet enough not to fill a 64 KiB pipe on its own,
|
||||
/// so this catches a regression to `wait()` only in combination with a verbose input. It
|
||||
/// is still worth pinning — a reverted drain plus any chatty clip is data loss.
|
||||
#[tokio::test]
|
||||
async fn a_real_clip_yields_a_poster_well_inside_the_timeout() {
|
||||
if tokio::process::Command::new("ffmpeg")
|
||||
.arg("-version")
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.await
|
||||
.is_err()
|
||||
{
|
||||
eprintln!("skipping: ffmpeg not on PATH");
|
||||
return;
|
||||
}
|
||||
let src = Path::new("../e2e/fixtures/media/sample.mp4");
|
||||
if !src.exists() {
|
||||
eprintln!("skipping: {} missing", src.display());
|
||||
return;
|
||||
}
|
||||
|
||||
let dir = std::env::temp_dir().join(format!("es-video-ok-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let dest = dir.join("poster.jpg");
|
||||
|
||||
let started = std::time::Instant::now();
|
||||
let got = extract_poster_frame(src, &dest, 400).await;
|
||||
let elapsed = started.elapsed();
|
||||
|
||||
assert!(matches!(got, Ok(true)), "expected a poster, got {got:?}");
|
||||
assert!(dest.metadata().unwrap().len() > 0);
|
||||
assert!(
|
||||
elapsed < FFMPEG_TIMEOUT / 4,
|
||||
"extraction took {elapsed:?}; a drained stderr finishes in well under \
|
||||
{FFMPEG_TIMEOUT:?} — this is the pipe-deadlock regression guard"
|
||||
);
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
}
|
||||
@@ -3,7 +3,10 @@ use tokio::sync::broadcast;
|
||||
|
||||
use crate::config::AppConfig;
|
||||
use crate::services::compression::CompressionWorker;
|
||||
use crate::services::config::ConfigCache;
|
||||
use crate::services::disk::DiskCache;
|
||||
use crate::services::rate_limiter::RateLimiter;
|
||||
use crate::services::sse_tickets::SseTicketStore;
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct SseEvent {
|
||||
@@ -11,6 +14,17 @@ pub struct SseEvent {
|
||||
pub data: String,
|
||||
}
|
||||
|
||||
impl SseEvent {
|
||||
/// Standardised constructor. Prefer this over building the struct inline so the
|
||||
/// event-type strings stay consistent across handlers.
|
||||
pub fn new(event_type: impl Into<String>, data: impl Into<String>) -> Self {
|
||||
Self {
|
||||
event_type: event_type.into(),
|
||||
data: data.into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct AppState {
|
||||
pub pool: PgPool,
|
||||
@@ -18,19 +32,37 @@ pub struct AppState {
|
||||
pub sse_tx: broadcast::Sender<SseEvent>,
|
||||
pub compression: CompressionWorker,
|
||||
pub rate_limiter: RateLimiter,
|
||||
pub sse_tickets: SseTicketStore,
|
||||
/// In-memory cache in front of the `config` table. Reads go through here; the
|
||||
/// admin PATCH handler and the test reseed invalidate it after committing.
|
||||
pub config_cache: ConfigCache,
|
||||
/// Cached total/free bytes for the media filesystem (quota + admin stats).
|
||||
pub disk_cache: DiskCache,
|
||||
}
|
||||
|
||||
impl AppState {
|
||||
pub fn new(pool: PgPool, config: AppConfig) -> Self {
|
||||
let (sse_tx, _) = broadcast::channel(256);
|
||||
let compression =
|
||||
CompressionWorker::new(pool.clone(), config.media_path.clone(), 2);
|
||||
// Broadcast buffer for live SSE fan-out. Sized to absorb a burst (e.g. many
|
||||
// uploads landing at once during a busy moment) before a slow consumer lags and
|
||||
// has to `resync`. The resync path is a correctness backstop, not the happy path —
|
||||
// a roomier buffer keeps ~1000 concurrent clients from all resyncing at once.
|
||||
let (sse_tx, _) = broadcast::channel(1024);
|
||||
let compression = CompressionWorker::new(
|
||||
pool.clone(),
|
||||
config.media_path.clone(),
|
||||
config.compression_concurrency,
|
||||
sse_tx.clone(),
|
||||
);
|
||||
let config_cache = ConfigCache::new(pool.clone());
|
||||
Self {
|
||||
pool,
|
||||
config,
|
||||
sse_tx,
|
||||
compression,
|
||||
rate_limiter: RateLimiter::new(),
|
||||
sse_tickets: SseTicketStore::new(),
|
||||
config_cache,
|
||||
disk_cache: DiskCache::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
export const env={}
|
||||
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
import{u as o,n as t,o as c}from"./CcONa1Mr.js";function u(e){throw new Error("https://svelte.dev/e/lifecycle_outside_component")}function r(e){t===null&&u(),o(()=>{const n=c(e);if(typeof n=="function")return n})}export{r as o};
|
||||
@@ -1 +0,0 @@
|
||||
import{f as l,g as o,p as u,i as n,j as d,k as m,h as p,e as _,m as v,l as k}from"./CcONa1Mr.js";class w{anchor;#t=new Map;#s=new Map;#e=new Map;#i=new Set;#f=!0;constructor(t,s=!0){this.anchor=t,this.#f=s}#a=t=>{if(this.#t.has(t)){var s=this.#t.get(t),e=this.#s.get(s);if(e)l(e),this.#i.delete(s);else{var f=this.#e.get(s);f&&(this.#s.set(s,f.effect),this.#e.delete(s),f.fragment.lastChild.remove(),this.anchor.before(f.fragment),e=f.effect)}for(const[i,a]of this.#t){if(this.#t.delete(i),i===t)break;const r=this.#e.get(a);r&&(o(r.effect),this.#e.delete(a))}for(const[i,a]of this.#s){if(i===s||this.#i.has(i))continue;const r=()=>{if(Array.from(this.#t.values()).includes(i)){var c=document.createDocumentFragment();v(a,c),c.append(n()),this.#e.set(i,{effect:a,fragment:c})}else o(a);this.#i.delete(i),this.#s.delete(i)};this.#f||!e?(this.#i.add(i),u(a,r,!1)):r()}}};#r=t=>{this.#t.delete(t);const s=Array.from(this.#t.values());for(const[e,f]of this.#e)s.includes(e)||(o(f.effect),this.#e.delete(e))};ensure(t,s){var e=m,f=k();if(s&&!this.#s.has(t)&&!this.#e.has(t))if(f){var i=document.createDocumentFragment(),a=n();i.append(a),this.#e.set(t,{effect:d(()=>s(a)),fragment:i})}else this.#s.set(t,d(()=>s(this.anchor)));if(this.#t.set(e,t),f){for(const[r,h]of this.#s)r===t?e.unskip_effect(h):e.skip_effect(h);for(const[r,h]of this.#e)r===t?e.unskip_effect(h.effect):e.skip_effect(h.effect);e.oncommit(this.#a),e.ondiscard(this.#r)}else p&&(this.anchor=_),this.#a(e)}}export{w as B};
|
||||
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
import{b as c,h as o,a as l,E as b,r as p,s as v,c as g,d,e as m}from"./CcONa1Mr.js";import{B as y}from"./BRDva_z9.js";function k(f,h,_=!1){var n;o&&(n=m,l());var s=new y(f),u=_?b:0;function t(a,r){if(o){var e=p(n);if(a!==parseInt(e.substring(1))){var i=v();g(i),s.anchor=i,d(!1),s.ensure(a,r),d(!0);return}}s.ensure(a,r)}c(()=>{var a=!1;h((r,e=0)=>{a=!0,t(e,r)}),a||t(-1,null)},u)}export{k as i};
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
import{A as v,i as d,B as l,C as u,D as T,T as p,F as h,h as i,e as s,R as E,a as y,G as g,c as w,H as N}from"./CcONa1Mr.js";const A=globalThis?.window?.trustedTypes&&globalThis.window.trustedTypes.createPolicy("svelte-trusted-html",{createHTML:t=>t});function M(t){return A?.createHTML(t)??t}function x(t){var r=v("template");return r.innerHTML=M(t.replaceAll("<!>","<!---->")),r.content}function n(t,r){var e=l;e.nodes===null&&(e.nodes={start:t,end:r,a:null,t:null})}function b(t,r){var e=(r&p)!==0,f=(r&h)!==0,a,_=!t.startsWith("<!>");return()=>{if(i)return n(s,null),s;a===void 0&&(a=x(_?t:"<!>"+t),e||(a=u(a)));var o=f||T?document.importNode(a,!0):a.cloneNode(!0);if(e){var c=u(o),m=o.lastChild;n(c,m)}else n(o,o);return o}}function C(t=""){if(!i){var r=d(t+"");return n(r,r),r}var e=s;return e.nodeType!==g?(e.before(e=d()),w(e)):N(e),n(e,e),e}function O(){if(i)return n(s,null),s;var t=document.createDocumentFragment(),r=document.createComment(""),e=d();return t.append(r,e),n(r,e),t}function P(t,r){if(i){var e=l;((e.f&E)===0||e.nodes.end===null)&&(e.nodes.end=s),y();return}t!==null&&t.before(r)}const L="5";typeof window<"u"&&((window.__svelte??={}).v??=new Set).add(L);export{P as a,n as b,O as c,b as f,C as t};
|
||||
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
import{l as o,a as r}from"../chunks/Dy1jDy4J.js";export{o as load_css,r as start};
|
||||
@@ -1 +0,0 @@
|
||||
import{c as s,a as c}from"../chunks/RsTAN2PN.js";import{b as l,E as p,t as i}from"../chunks/CcONa1Mr.js";import{B as m}from"../chunks/BRDva_z9.js";function u(n,r,...e){var o=new m(n);l(()=>{const t=r()??null;o.ensure(t,t&&(a=>t(a,...e)))},p)}const f=!0,_=!1,g=Object.freeze(Object.defineProperty({__proto__:null,prerender:f,ssr:_},Symbol.toStringTag,{value:"Module"}));function h(n,r){var e=s(),o=i(e);u(o,()=>r.children),c(n,e)}export{h as component,g as universal};
|
||||
@@ -1 +0,0 @@
|
||||
import{a as i,f as h}from"../chunks/RsTAN2PN.js";import{q as g,t as v,v as d,w as l,x as s,y as a,z as x}from"../chunks/CcONa1Mr.js";import{s as o}from"../chunks/Bb9JxzU7.js";import{s as _,p}from"../chunks/Dy1jDy4J.js";const $={get error(){return p.error},get status(){return p.status}};_.updated.check;const m=$;var k=h("<h1> </h1> <p> </p>",1);function z(c,f){g(f,!0);var t=k(),r=v(t),n=s(r,!0);a(r);var e=x(r,2),u=s(e,!0);a(e),d(()=>{o(n,m.status),o(u,m.error?.message)}),i(c,t),l()}export{z as component};
|
||||
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
{"version":"1775501323159"}
|
||||
File diff suppressed because one or more lines are too long
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user