Compare commits
146 Commits
be493649af
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
33cc41bacd | ||
|
|
08a9819d76 | ||
|
|
70a6598924 | ||
|
|
e6aefaa804 | ||
|
|
6354a7a8a3 | ||
|
|
60d5a2efea | ||
|
|
90f2398e56 | ||
|
|
e960aae163 | ||
|
|
adce950049 | ||
|
|
bde72e47f4 | ||
|
|
cb757e7b69 | ||
|
|
042e7e9047 | ||
|
|
cc8ae2566f | ||
|
|
4f085f228a | ||
|
|
ac5641bcd3 | ||
|
|
cd5358dc9b | ||
|
|
c31830468c | ||
|
|
75f4481bbd | ||
|
|
958d5bd713 | ||
|
|
f2c9cfe162 | ||
|
|
b5f7467c47 | ||
|
|
9148e23da8 | ||
|
|
f39307232c | ||
|
|
1b7b8a3038 | ||
|
|
253d46c7e5 | ||
|
|
c570e0cc37 | ||
|
|
5784483a57 | ||
|
|
a44511983d | ||
|
|
f8e53809d5 | ||
|
|
0afd202164 | ||
|
|
ba3e5b481b | ||
|
|
3b783c1d9b | ||
|
|
a47b6895c2 | ||
|
|
3ca05dcb58 | ||
|
|
c308eb3eac | ||
|
|
69a9309c54 | ||
|
|
1ed1a134ea | ||
|
|
32d0a7e13b | ||
|
|
d779fa2b97 | ||
|
|
755417730f | ||
|
|
5b1ce581f3 | ||
|
|
c11df3182c | ||
|
|
d749b56d58 | ||
|
|
c24e296f07 | ||
|
|
9508fb8e86 | ||
|
|
ca55712622 | ||
|
|
4fe435cc76 | ||
|
|
f5cb460aec | ||
|
|
f83d49b83e | ||
|
|
cef41ce76a | ||
|
|
7570524e5b | ||
|
|
5b46eab73a | ||
|
|
91d6a50dba | ||
| 134ab54b34 | |||
|
|
f879ce1866 | ||
|
|
bf425cf8e6 | ||
|
|
ff4ca964f5 | ||
|
|
61669aac3f | ||
|
|
795eb76f9a | ||
|
|
ebf0b8289b | ||
|
|
ce9a727c73 | ||
|
|
46134c8760 | ||
|
|
b7d8faadf7 | ||
|
|
b0500e8e48 | ||
|
|
379224bee4 | ||
|
|
1c955458d6 | ||
|
|
987d1ba235 | ||
|
|
c69ff502f7 | ||
|
|
51ea254dde | ||
|
|
fee68dd9ac | ||
|
|
3c8e264f48 | ||
|
|
16cdec051a | ||
|
|
5b51bcf056 | ||
|
|
f1e66141f4 | ||
|
|
cee1e73f98 | ||
|
|
ec73c6e001 | ||
|
|
bd6ae86a85 | ||
|
|
3622dcc02f | ||
|
|
9cb2f152d3 | ||
|
|
7c3c9cf699 | ||
|
|
0d9505ce9f | ||
|
|
7bdbe3ce5b | ||
|
|
3364ea52c9 | ||
|
|
def97d3087 | ||
|
|
09f12c8959 | ||
|
|
34c1122e4e | ||
|
|
200ab1c0a0 | ||
|
|
ef8d226ba6 | ||
|
|
19db66a845 | ||
|
|
d6a109df2d | ||
|
|
bb833c7e71 | ||
|
|
258a536254 | ||
|
|
9c4a93c058 | ||
|
|
bfaa166e0a | ||
|
|
3824eaafb2 | ||
|
|
f1349100d7 | ||
|
|
3f1a5e9c41 | ||
|
|
c212deb7b0 | ||
|
|
f5692ea109 | ||
|
|
ad1689b818 | ||
|
|
01d18e7ba2 | ||
|
|
2267a83f6c | ||
|
|
f318c3bf51 | ||
|
|
cf8971faae | ||
|
|
38146b4d03 | ||
|
|
8acc0e6cc2 | ||
|
|
4e154434a1 | ||
|
|
141cd52f7e | ||
|
|
83a9ab40cd | ||
|
|
a141d65db1 | ||
|
|
3a9e7ca2da | ||
|
|
5dc93bfb84 | ||
|
|
592747f1e0 | ||
|
|
a0d63ac9fd | ||
|
|
35c02066fe | ||
|
|
ded77fe4ba | ||
|
|
4d02b56b77 | ||
|
|
0865659ab3 | ||
|
|
5596e1920d | ||
|
|
3cba9ecf95 | ||
|
|
ed18d95bb0 | ||
|
|
2ed42f7b9e | ||
|
|
cbbc626768 | ||
|
|
a3e53f303b | ||
|
|
2af421893f | ||
|
|
48f0439273 | ||
|
|
de7aefef69 | ||
|
|
3ba30f4ba9 | ||
|
|
4306d1c96a | ||
|
|
886caaecfa | ||
|
|
de510cc19a | ||
|
|
5b76e0cc37 | ||
|
|
66ae4b221b | ||
|
|
83c2899373 | ||
|
|
4fb98e4a1e | ||
|
|
5130c9933e | ||
|
|
e9331747d0 | ||
|
|
af6a07bd1f | ||
|
|
5e8323d7fa | ||
|
|
2715275065 | ||
|
|
198a1d46b0 | ||
|
|
3a36796768 | ||
|
|
bd7c3fc28c | ||
|
|
4456deb146 | ||
|
|
94591dfda4 | ||
|
|
194adf1339 |
85
.env.example
85
.env.example
@@ -25,6 +25,12 @@ DATABASE_URL=postgres://mangalord:mangalord@postgres:5432/mangalord
|
||||
BIND_ADDRESS=0.0.0.0:8080
|
||||
STORAGE_DIR=/var/lib/mangalord/storage
|
||||
RUST_LOG=info,mangalord=debug,chromiumoxide::conn=off,chromiumoxide::handler=off
|
||||
# Postgres connection-pool sizing. One pool serves HTTP handlers and the
|
||||
# crawler/analysis daemons. DB_MAX_CONNECTIONS caps open connections;
|
||||
# DB_ACQUIRE_TIMEOUT_SECS is how long a request waits for a free connection
|
||||
# before failing fast (rather than hanging on the driver's 30s default).
|
||||
DB_MAX_CONNECTIONS=20
|
||||
DB_ACQUIRE_TIMEOUT_SECS=10
|
||||
|
||||
# ----- Auth / cookies -----
|
||||
# COOKIE_SECURE controls whether the `Secure` flag is set on the session
|
||||
@@ -45,6 +51,13 @@ SESSION_TTL_DAYS=30
|
||||
# rate-limiting reverse proxy that already enforces a budget).
|
||||
AUTH_RATE_PER_SEC=5
|
||||
AUTH_RATE_BURST=10
|
||||
# Trust a proxy-supplied X-Forwarded-For as the client IP for per-IP auth
|
||||
# rate limiting. Enable ONLY when the backend sits behind a trusted proxy
|
||||
# that overrides the header (the compose deploy: SvelteKit forwards the real
|
||||
# peer IP). When false, the header is ignored and a single shared bucket is
|
||||
# used — a directly-exposed backend MUST keep this off or clients could spoof
|
||||
# their IP to dodge the limit.
|
||||
AUTH_TRUSTED_PROXY=false
|
||||
|
||||
# ----- CORS -----
|
||||
# Comma-separated origins allowed to call the API with credentials.
|
||||
@@ -62,11 +75,16 @@ CORS_ALLOWED_ORIGINS=
|
||||
# neither Origin nor Referer (curl, server-to-server callers) are
|
||||
# always allowed.
|
||||
#
|
||||
# Default is empty: CSRF check disabled (operator opt-out). For a
|
||||
# browser-exposed deployment this should be set to the SvelteKit
|
||||
# origin, e.g. https://app.example.com. For a same-origin
|
||||
# docker-compose deploy where only one origin exists, set the same
|
||||
# value the browser uses.
|
||||
# Empty does NOT mean "off" for browsers: cookie-authenticated admin
|
||||
# mutations FAIL CLOSED (403) when this is empty, since there's no
|
||||
# allowlist to check the Origin against. Non-cookie callers (curl,
|
||||
# bots with a Bearer token, no Origin/Referer) are still allowed.
|
||||
# So set this to the SvelteKit origin for any browser-exposed deploy,
|
||||
# e.g. https://app.example.com. For a same-origin docker-compose deploy
|
||||
# set the same value the browser uses.
|
||||
# Local dev (native `npm run dev`): the Vite origin is
|
||||
# http://localhost:5173 — set ADMIN_ALLOWED_ORIGINS=http://localhost:5173
|
||||
# or admin toggles (e.g. enabling the analysis worker) return 403.
|
||||
ADMIN_ALLOWED_ORIGINS=
|
||||
|
||||
# ----- Admin bootstrap -----
|
||||
@@ -98,6 +116,13 @@ MAX_REQUEST_BYTES=209715200
|
||||
# oversized image is rejected even when the total request fits.
|
||||
# Default 20 MiB.
|
||||
MAX_FILE_BYTES=20971520
|
||||
# Max page images accepted in one chapter upload. Bounds how many parts
|
||||
# the handler will stage before rejecting the request with 413, so a
|
||||
# client can't pin a worker with an unbounded page count. Default 2000.
|
||||
# Setting 0 disables THIS cap — the total is then bounded only by
|
||||
# MAX_REQUEST_BYTES (the whole-request body limit above), so leave 0 only
|
||||
# if you intend that body limit to be the sole backstop.
|
||||
MAX_PAGES_PER_CHAPTER=2000
|
||||
|
||||
# ----- Crawler download safety -----
|
||||
# Hosts the crawler is allowed to fetch images/covers from, in addition
|
||||
@@ -112,6 +137,11 @@ CRAWLER_DOWNLOAD_ALLOWLIST=
|
||||
CRAWLER_ALLOW_ANY_HOST=false
|
||||
# Hard cap on a single image body. Default 32 MiB.
|
||||
CRAWLER_MAX_IMAGE_BYTES=33554432
|
||||
# Hard cap on the number of page images in one crawled chapter. The
|
||||
# per-image byte cap doesn't stop a hostile reader page listing thousands
|
||||
# of <img> tags; an over-cap chapter is acked failed instead of downloaded.
|
||||
# 0 disables the cap. Default 2000.
|
||||
CRAWLER_MAX_IMAGES_PER_CHAPTER=2000
|
||||
# Max manga detail fetches per metadata pass (both the in-process daemon
|
||||
# and the `bin/crawler` CLI). 0 means no cap — let the source walker run
|
||||
# to completion. Useful for capped test runs against a new source.
|
||||
@@ -163,6 +193,19 @@ CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES=10
|
||||
# exhausted) that trigger an automatic coordinated browser restart.
|
||||
# Default 3.
|
||||
CRAWLER_BROWSER_RESTART_THRESHOLD=3
|
||||
# CDP Fetch interception that re-validates every headless-browser
|
||||
# navigation/redirect/subresource against the SSRF check (blocks a scraped page
|
||||
# that drives the browser — via a redirect OR page JS/subresource — to an
|
||||
# internal target such as the cloud metadata service or postgres). Default TRUE:
|
||||
# with it off, only the top-level URL string is checked and page JS/subresources
|
||||
# reach internal IPs (the reqwest SafeResolver does not cover Chromium's own
|
||||
# stack). The CDP Fetch hook can't be exercised in CI (no Chromium); before
|
||||
# relying on a fresh deploy, validate it doesn't wedge navigation:
|
||||
# CRAWLER_CHROMIUM_BINARY=/usr/bin/chromium \
|
||||
# cargo test -p mangalord --test crawler_browser_smoke -- --ignored \
|
||||
# ssrf_interception_does_not_wedge_allowed_navigation
|
||||
# Set to false only as a break-glass if the hook destabilizes a deployment.
|
||||
CRAWLER_SSRF_INTERCEPT=true
|
||||
# Path to a system Chromium binary. When set, the crawler skips the
|
||||
# bundled-fetcher download. Required on platforms without a usable
|
||||
# upstream Chromium build (notably Linux_arm64 / Raspberry Pi). On
|
||||
@@ -277,6 +320,22 @@ CRAWLER_TZ=UTC
|
||||
# ANALYSIS_ENABLED Turn the worker on at first boot. Toggleable live in
|
||||
# the dashboard. Default `false`.
|
||||
ANALYSIS_ENABLED=false
|
||||
# ANALYSIS_BACKEND Which engine the worker runs. env-ONLY (deploy-time).
|
||||
# `ocr` (default) = the in-process ocrs engine: fast,
|
||||
# CPU-only, text-only, ideal for a Pi.
|
||||
# NOTE: the `vision` backend (local LLM at ANALYSIS_VISION_URL,
|
||||
# full OCR + tags + scene + safety) is TEMPORARILY DISABLED —
|
||||
# its code is kept intact but the worker forces OCR regardless,
|
||||
# logging a warning if `vision` is requested. So setting
|
||||
# `vision` here currently has no effect; the ANALYSIS_VISION_* /
|
||||
# ANALYSIS_API_KEY knobs below are dormant until it's re-enabled.
|
||||
ANALYSIS_BACKEND=ocr
|
||||
# OCRS_DETECTION_MODEL / OCRS_RECOGNITION_MODEL Paths to the ocrs `.rten`
|
||||
# text-detection / -recognition models (only read when ANALYSIS_BACKEND=ocr).
|
||||
# The backend image bakes both into /models, so the defaults work unchanged;
|
||||
# override only to point at custom-trained models.
|
||||
OCRS_DETECTION_MODEL=/models/text-detection.rten
|
||||
OCRS_RECOGNITION_MODEL=/models/text-recognition.rten
|
||||
# ANALYSIS_VISION_URL /v1/chat/completions endpoint. Required when enabled.
|
||||
# For the bundled vision container, use
|
||||
# http://mangalord-vision:8000/v1/chat/completions.
|
||||
@@ -307,6 +366,10 @@ ANALYSIS_API_KEY=
|
||||
# beyond cap). Default 16.
|
||||
# ANALYSIS_MAX_IMAGE_BYTES Per-image byte cap before downscale.
|
||||
# Default 8388608 (8 MiB).
|
||||
# ANALYSIS_OCR_MAX_DECODE_PIXELS Decompression-bomb guard for the OCR
|
||||
# backend: hard cap on a page's *decoded* pixel
|
||||
# count (encoded size is bounded separately).
|
||||
# Default 100000000 (100 MP).
|
||||
# ANALYSIS_RESPONSE_FORMAT `json_schema` | `json_object` | `none`.
|
||||
# Default `json_schema`.
|
||||
# ANALYSIS_FREQUENCY_PENALTY Discourages repetition loops. Default 0.3.
|
||||
@@ -319,6 +382,7 @@ ANALYSIS_SLICE_OVERLAP=0.12
|
||||
ANALYSIS_TALL_ASPECT=1.6
|
||||
ANALYSIS_MAX_SLICES=16
|
||||
ANALYSIS_MAX_IMAGE_BYTES=8388608
|
||||
ANALYSIS_OCR_MAX_DECODE_PIXELS=100000000
|
||||
ANALYSIS_RESPONSE_FORMAT=json_schema
|
||||
ANALYSIS_FREQUENCY_PENALTY=0.3
|
||||
ANALYSIS_TEMPERATURE=0.0
|
||||
@@ -363,3 +427,14 @@ VISION_MANAGER_DATABASE_URL=
|
||||
# VISION_MEM_LOW_WATERMARK_PCT used% >= this → inhibit starts (keep HIGH - LOW >= ~10 so the band beats jitter). Default 80.
|
||||
# VISION_MEM_YIELD_COOLDOWN Seconds after a pressure-stop during which restart is refused regardless of backlog. Default 300.
|
||||
# VISION_MEM_POLL_INTERVAL Mem sub-poll cadence, seconds (<= VISION_POLL_INTERVAL); catches spikes between backlog ticks. Default 5.
|
||||
|
||||
# ---- Deployment resource bounds & exposure (docker-compose.yml) ----
|
||||
# Host interface the frontend port publishes on. Default 127.0.0.1 so SvelteKit
|
||||
# (plain HTTP) is reachable only by a host-local reverse proxy. Set 0.0.0.0 only
|
||||
# if your TLS terminator runs on a different host.
|
||||
FRONTEND_PUBLISH_ADDR=127.0.0.1
|
||||
# Per-container memory ceilings (docker mem_limit). Generous defaults; tune to
|
||||
# your host. The backend runs OCR + headless Chromium and is the heavy one.
|
||||
BACKEND_MEM_LIMIT=4g
|
||||
FRONTEND_MEM_LIMIT=512m
|
||||
POSTGRES_MEM_LIMIT=1g
|
||||
|
||||
@@ -136,8 +136,8 @@ docker compose -f docker-compose.dev.yml up -d
|
||||
These are first-class slots in the architecture. When adding any of them, plug into the existing seam rather than building parallel infrastructure.
|
||||
|
||||
- **Tags / lists**: new tables joined to `mangas`. New `domain`, `repo`, and `api` modules; the existing manga endpoints do not need to change.
|
||||
- **Per-page collections / tags**: `collections` is heterogeneous — `collection_mangas` holds whole mangas, `collection_pages` holds individual pages (FK to `pages.id`). Per-user page tags live in `page_tags`, which references the **shared** `tags` table by `tag_id` (migration 0024) — the same lookup table `manga_tags` uses, so manga tags and page tags share one global vocabulary. The HTTP contract still speaks tag *names*; `repo::page_tag` resolves name↔id via `repo::tag::upsert_by_name` and applies the stricter page-tag `normalize_tag` (lowercase, collapse whitespace, reject wildcards/control/invisible chars) at the API layer. Both `collection_pages` and `page_tags` cascade-delete with `pages` and `chapters`, so re-uploading a chapter drops saved-page references by design.
|
||||
- **Tag-based content search (`/search`)**: the user-facing search surface lives at [frontend/src/routes/search/+page.svelte](frontend/src/routes/search/+page.svelte). Three result views (Pages / Chapters / Mangas) consume the matching `/v1/me/page-tags`, `/v1/me/page-tags/chapters`, and `/v1/me/page-tags/mangas` endpoints. Note the two distinct query-param spaces: `?q=` on `/v1/me/page-tags` is a tag-name prefix (for autocomplete in the "Add tag" sheet); `?text=` on the aggregation endpoints is **reserved** for the planned OCR text-search input. Both aggregation handlers accept `text=` on the wire but reject non-empty values with 501 `text_search_not_yet_supported` so adding OCR later doesn't break the API shape. Adding OCR is then: a background worker writes `page_ocr_text` rows, a JOIN on the existing aggregation queries adds the new filter, the `text=` param starts validating instead of rejecting.
|
||||
- **Per-page collections / tags**: `collections` is heterogeneous — `collection_mangas` holds whole mangas, `collection_pages` holds individual pages (FK to `pages.id`). Per-user page tags live in `page_tags`, which references the **shared** `tags` table by `tag_id` (migration 0024) — the same lookup table `manga_tags` uses, so manga tags and page tags share one global vocabulary. The HTTP contract still speaks tag *names*; `repo::page_tag` resolves name↔id via `repo::tag::upsert_by_name` and applies the stricter page-tag `normalize_tag` (lowercase, collapse whitespace, reject wildcards/control/invisible chars) at the API layer. Both `collection_pages` and `page_tags` cascade-delete with `pages` and `chapters`, so a saved-page reference is dropped only when its `pages` row is genuinely deleted — i.e. when the chapter is deleted (cascade), or when a user deletes and re-creates a chapter (the user upload path in `api::chapters::finalize_chapter` inserts a *new* chapter row with fresh page ids). A crawler **re-fetch** does **not** drop saves: `content::persist_pages` upserts pages by `(chapter_id, page_number)` (`ON CONFLICT (…) DO UPDATE … RETURNING id`), preserving each `pages.id`, so collections and page tags keyed on that id survive the re-fetch. (Migration 0023's header comment predates this and describes the cascade as an unconditional "re-upload drops saves"; the checked-in migration text is intentionally left as-is because sqlx checksums applied migrations.)
|
||||
- **Tag-based content search (`/search`)**: the user-facing search surface lives at [frontend/src/routes/search/+page.svelte](frontend/src/routes/search/+page.svelte). Three result views (Pages / Chapters / Mangas) consume the matching `/v1/me/page-tags`, `/v1/me/page-tags/chapters`, and `/v1/me/page-tags/mangas` endpoints. Note the two distinct query-param spaces: `?q=` on `/v1/me/page-tags` is a tag-name prefix (for autocomplete in the "Add tag" sheet); `?text=` on the aggregation endpoints performs **OCR full-text search** — the active OCR backend writes `page_ocr_text` rows and a weighted `search_doc` tsvector, and the aggregation queries JOIN on a `plainto_tsquery` filter ranked by `ts_rank` (see [backend/src/repo/page_analysis.rs](backend/src/repo/page_analysis.rs)). (`text=` was previously reserved and returned 501 `text_search_not_yet_supported`; that placeholder is gone now that the OCR backend is active. The generic `AppError::NotImplemented` 501 mechanism remains for future feature reservations.)
|
||||
- **Full-text / fuzzy search**: enable `pg_trgm` in a migration and add a GIN index on `mangas.title`; swap the `WHERE` in `repo::manga::list` to use `%` operator or `tsvector`. The API shape (`?search=...`) does not change.
|
||||
- **OCR / autotagging**: a background worker (a separate binary or a tokio task spawned in `app::build`) that reads pages from `storage::Storage` and writes tag rows. Do not couple OCR to upload handlers — it runs asynchronously.
|
||||
- **S3 storage**: add `storage::S3Storage` implementing `Storage`. Branch in `app::build` based on a config field (e.g., `STORAGE_BACKEND=s3`). Handlers do not change.
|
||||
|
||||
136
HYBRID-OCR-VISION.md
Normal file
136
HYBRID-OCR-VISION.md
Normal file
@@ -0,0 +1,136 @@
|
||||
# Hybrid OCR→Vision analysis pipeline — design brief for the dev agent
|
||||
|
||||
**Status: design only (not implemented).** Forward-looking model for a future
|
||||
`ANALYSIS_BACKEND=hybrid`, on top of the shipped `ocr` (default) and `vision`
|
||||
backends. Constraint that shapes it: `ocrs` and the vision LLM must **not** be
|
||||
resident at the same time (Pi RAM ceiling). OCR has priority; vision is a
|
||||
preemptible, only-when-OCR-is-drained background phase.
|
||||
|
||||
## The key realization: this is the crawl mutex, again
|
||||
|
||||
The repo already runs a **`vision-manager`** sidecar
|
||||
([vision-manager/manager.sh](vision-manager/manager.sh),
|
||||
[VISION-AUTOSCALE.md](VISION-AUTOSCALE.md)) that owns the `mangalord-vision`
|
||||
(llama.cpp, ~4 GiB) container lifecycle via a scoped docker-socket-proxy — the
|
||||
backend gets no Docker access. It already:
|
||||
|
||||
- **Starts** vision when `pending_analysis()` > 0 (counts `analyze_page` jobs).
|
||||
- **Idle-stops** it after `STOP_DEBOUNCE` (anti-thrash) once the queue drains.
|
||||
- **Defers starting** vision while `crawl_running()` > 0 — the **"RAM mutex"**:
|
||||
the crawler's browser and the LLM can't both fit, so vision waits for crawl.
|
||||
- **Memory-yield**: SIGTERMs a running vision at a HIGH host-RAM watermark,
|
||||
blocks starting at a LOW watermark (see [VISION-MEMORY-YIELD.md](VISION-MEMORY-YIELD.md)).
|
||||
|
||||
The OCR↔vision relationship is the relationship that already exists between
|
||||
crawl and vision. OCR is just a third RAM-competing, higher-priority workload
|
||||
that vision must defer to. Extend the existing arbiter rather than build a new one.
|
||||
|
||||
## Job model: split one queue into two
|
||||
|
||||
Today there is one job kind, `analyze_page`. Split it:
|
||||
|
||||
- **`ocr_page`** — the in-process ocrs pass (cheap, fast, no container).
|
||||
- **`ground_page`** — the vision grounding pass (needs the LLM container).
|
||||
|
||||
**Pipeline (hybrid mode):** upload → enqueue `ocr_page`. The OCR daemon runs
|
||||
ocrs, writes `page_ocr_text` + `search_doc` (text searchable *immediately*), then
|
||||
enqueues `ground_page` for that page. The grounding daemon later adds
|
||||
tags/scene/safety. Each kind gets its own dedup index (mirror migration 0031).
|
||||
|
||||
Two daemons, two queues.
|
||||
|
||||
## Arbitration: OCR priority + memory exclusivity
|
||||
|
||||
Responsibilities split cleanly between the in-backend daemons and the manager:
|
||||
|
||||
**Backend — OCR daemon** (already built): leases `ocr_page`, always allowed to
|
||||
run (ocrs is small/local). This is the priority phase.
|
||||
|
||||
**Backend — grounding daemon:** leases `ground_page`, but gates leasing on
|
||||
**both**:
|
||||
1. `VisionReadiness` (the existing `/health` gate — vision container up), AND
|
||||
2. **`ocr_backlog == 0`** (a cheap `count(ocr_page WHERE state IN pending,running)`).
|
||||
|
||||
Gate #2 is the whole priority mechanism, in-process, no aborts: the instant OCR
|
||||
work appears, the grounding daemon **finishes its current page** (the in-flight
|
||||
lease completes normally) and then **stops leasing** new pages and parks — "it
|
||||
finishes, then yields." It resumes only when OCR has fully drained.
|
||||
|
||||
**vision-manager:** two one-line changes to the existing logic:
|
||||
1. `pending_analysis()` counts **`ground_page`** (not `ocr_page`) — vision only
|
||||
starts when there is *grounding* work.
|
||||
2. Add an **OCR start-block mutex** identical to `RESPECT_CRAWL_MUTEX`: defer
|
||||
starting vision while `ocr_page` backlog > 0. (`RESPECT_OCR_MUTEX=1`.)
|
||||
|
||||
**Why this yields memory exclusivity:** while OCR backlog > 0, the grounding
|
||||
daemon won't lease → vision goes idle → the manager's idle-debounce stops the
|
||||
container (and never restarts it under the OCR mutex). Only ocrs (small) is
|
||||
resident. When OCR drains, grounding resumes, `pending_analysis()` > 0 again, the
|
||||
manager starts vision. Only the *big* consumer (the LLM container) is ever
|
||||
mutually exclusive with active OCR — which is the actual RAM constraint.
|
||||
|
||||
**The one overlap window:** if OCR work arrives mid-grounding-page, ocrs (small,
|
||||
in-process) briefly coexists with the one in-flight vision page before the
|
||||
grounding daemon parks. ocrs's footprint makes this a non-issue in practice. If a
|
||||
deployment needs *hard* exclusivity even there, the OCR daemon can additionally
|
||||
wait for `vision_running == false` before its first dispatch. No deadlock: OCR's
|
||||
wait is transient (one grounding page, bounded by `job_timeout`), while vision's
|
||||
deferral on OCR backlog is the persistent, priority-respecting side.
|
||||
|
||||
**Anti-thrash:** the manager's `STOP_DEBOUNCE` already prevents a stray OCR page
|
||||
from cold-cycling the 4 GiB model. Chapter uploads arrive as bursts, so OCR
|
||||
drains a whole batch in one phase before vision resumes — the natural good case.
|
||||
|
||||
## Data-model split (the real refactor)
|
||||
|
||||
`persist_analysis` currently writes OCR + tags + scene + safety in one
|
||||
transaction and derives `search_doc` from the OCR rows (+ scene weight C). Two
|
||||
passes means splitting it, sharing the tsvector builder:
|
||||
|
||||
- **`persist_ocr(page, lines)`** — writes `page_ocr_text`, sets
|
||||
`search_doc` from OCR (A/B/D buckets), marks the page text-searchable. Status
|
||||
reflects OCR completion.
|
||||
- **`persist_grounding(page, tags, scene, safety)`** — writes auto-tags,
|
||||
warnings, scene; **recomputes** `search_doc` to fold in the scene (weight C) on
|
||||
top of the existing OCR buckets. Records grounding completion (e.g. a
|
||||
`grounded_at` column or tags-present sentinel).
|
||||
|
||||
This keeps text search live after the cheap OCR phase, with semantic search/tags
|
||||
arriving after the (deferred) grounding phase.
|
||||
|
||||
## Reusable pieces
|
||||
|
||||
- OCR side: the shipped `OcrsEngine` + `OcrAnalyzeDispatcher`
|
||||
([backend/src/analysis/ocr.rs](backend/src/analysis/ocr.rs)) — retarget to
|
||||
`ocr_page`, and on success enqueue `ground_page`.
|
||||
- Vision side: factor `VisionClient::ground(image, mime, ocr_text)` out of
|
||||
`analyze()`'s Pass-B block
|
||||
([backend/src/analysis/vision.rs](backend/src/analysis/vision.rs) ~240–261) —
|
||||
pure refactor, existing tests cover it. The grounding daemon is the existing
|
||||
analysis daemon retargeted to `ground_page` with the extra `ocr_backlog==0`
|
||||
lease gate.
|
||||
- Manager: `pending_analysis()` kind swap + `RESPECT_OCR_MUTEX` block, mirroring
|
||||
the crawl-mutex branch already at [vision-manager/manager.sh](vision-manager/manager.sh).
|
||||
|
||||
## Scope / sequencing
|
||||
|
||||
Larger than an inline hybrid dispatcher: job-kind split + dedup migrations,
|
||||
`persist_*` split + search_doc-builder extraction, pipeline enqueue (OCR→ground),
|
||||
grounding daemon lease gate, and the manager mutex. No new privileged surface
|
||||
(reuses the docker-socket-proxy). Suggested order:
|
||||
1. Split `persist_analysis` → `persist_ocr` / `persist_grounding` (+ tests).
|
||||
2. Split job kinds + dedup migrations; retarget the OCR daemon to `ocr_page`.
|
||||
3. `VisionClient::ground()` refactor + grounding daemon (`ground_page`, gated on
|
||||
`ocr_backlog==0`).
|
||||
4. OCR→ground enqueue on OCR success (hybrid mode only).
|
||||
5. vision-manager: `ground_page` counting + `RESPECT_OCR_MUTEX`.
|
||||
|
||||
Pure-`ocr` (shipped default) and `vision` (legacy) backends are unaffected; this
|
||||
is the `hybrid` backend's runtime model.
|
||||
|
||||
## Open decision
|
||||
|
||||
Hard exclusivity in the one overlap window (OCR daemon waits for `vision_running
|
||||
== false` before its first dispatch) — needed only if ocrs's few-hundred-MB
|
||||
footprint can't coexist with a single in-flight grounding page on the target
|
||||
Pi's RAM. Default: don't gate (rely on the grounding daemon parking fast).
|
||||
4
backend/.gitignore
vendored
4
backend/.gitignore
vendored
@@ -1,3 +1,7 @@
|
||||
/target
|
||||
/.sqlx
|
||||
.env
|
||||
|
||||
# Local OCR models for native dev (downloaded, not source)
|
||||
models/
|
||||
*.rten
|
||||
|
||||
259
backend/Cargo.lock
generated
259
backend/Cargo.lock
generated
@@ -214,6 +214,12 @@ version = "1.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06"
|
||||
|
||||
[[package]]
|
||||
name = "bitflags"
|
||||
version = "1.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a"
|
||||
|
||||
[[package]]
|
||||
name = "bitflags"
|
||||
version = "2.11.1"
|
||||
@@ -514,6 +520,25 @@ dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-deque"
|
||||
version = "0.8.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9dd111b7b7f7d55b72c0a6ae361660ee5853c9af73f70c3c2ef6858b950e2e51"
|
||||
dependencies = [
|
||||
"crossbeam-epoch",
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-epoch"
|
||||
version = "0.9.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e"
|
||||
dependencies = [
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-queue"
|
||||
version = "0.3.12"
|
||||
@@ -638,7 +663,7 @@ version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e0e367e4e7da84520dedcac1901e4da967309406d1e51017ae1abfb97adbd38"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"objc2",
|
||||
]
|
||||
|
||||
@@ -772,6 +797,16 @@ version = "0.1.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582"
|
||||
|
||||
[[package]]
|
||||
name = "flatbuffers"
|
||||
version = "24.12.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4f1baf0dbf96932ec9a3038d57900329c015b0bfb7b63d904f3bc27e2b02a096"
|
||||
dependencies = [
|
||||
"bitflags 1.3.2",
|
||||
"rustc_version",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.1.9"
|
||||
@@ -1059,6 +1094,12 @@ version = "0.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
||||
|
||||
[[package]]
|
||||
name = "hermit-abi"
|
||||
version = "0.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c"
|
||||
|
||||
[[package]]
|
||||
name = "hex"
|
||||
version = "0.4.3"
|
||||
@@ -1448,7 +1489,7 @@ version = "0.1.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e02f3bb43d335493c96bf3fd3a321600bf6bd07ed34bc64118e9293bdffea46c"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"libc",
|
||||
"plain",
|
||||
"redox_syscall 0.7.5",
|
||||
@@ -1517,7 +1558,7 @@ checksum = "c41e0c4fef86961ac6d6f8a82609f55f31b05e4fce149ac5710e439df7619ba4"
|
||||
|
||||
[[package]]
|
||||
name = "mangalord"
|
||||
version = "0.88.0"
|
||||
version = "0.128.25"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"argon2",
|
||||
@@ -1537,14 +1578,15 @@ dependencies = [
|
||||
"infer",
|
||||
"mime",
|
||||
"nix 0.29.0",
|
||||
"ocrs",
|
||||
"rand 0.8.6",
|
||||
"reqwest",
|
||||
"rten",
|
||||
"scraper",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2",
|
||||
"sqlx",
|
||||
"subtle",
|
||||
"sysinfo",
|
||||
"tempfile",
|
||||
"thiserror 1.0.69",
|
||||
@@ -1669,7 +1711,7 @@ version = "0.29.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "71e2746dc3a24dd78b3cfcb7be93368c6de9963d30f43a6a73998a9cf4b17b46"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"cfg-if",
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
@@ -1681,7 +1723,7 @@ version = "0.31.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"cfg-if",
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
@@ -1757,6 +1799,16 @@ dependencies = [
|
||||
"libm",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num_cpus"
|
||||
version = "1.17.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "91df4bbde75afed763b708b7eee1e8e7651e02d97f6d5dd763e89367e957b23b"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "objc2"
|
||||
version = "0.6.4"
|
||||
@@ -1772,7 +1824,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "73ad74d880bb43877038da939b7427bba67e9dd42004a18b809ba7d87cee241c"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"objc2",
|
||||
"objc2-foundation",
|
||||
]
|
||||
@@ -1793,7 +1845,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"dispatch2",
|
||||
"objc2",
|
||||
]
|
||||
@@ -1804,7 +1856,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e022c9d066895efa1345f8e33e584b9f958da2fd4cd116792e15e07e4720a807"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"dispatch2",
|
||||
"objc2",
|
||||
"objc2-core-foundation",
|
||||
@@ -1837,7 +1889,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0cde0dfb48d25d2b4862161a4d5fcc0e3c24367869ad306b0c9ec0073bfed92d"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"objc2",
|
||||
"objc2-core-foundation",
|
||||
"objc2-core-graphics",
|
||||
@@ -1855,7 +1907,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e3e0adef53c21f888deb4fa59fc59f7eb17404926ee8a6f59f5df0fd7f9f3272"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"block2",
|
||||
"libc",
|
||||
"objc2",
|
||||
@@ -1868,7 +1920,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "180788110936d59bab6bd83b6060ffdfffb3b922ba1396b312ae795e1de9d81d"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"objc2",
|
||||
"objc2-core-foundation",
|
||||
]
|
||||
@@ -1879,7 +1931,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "96c1358452b371bf9f104e21ec536d37a650eb10f7ee379fff67d2e08d537f1f"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"objc2",
|
||||
"objc2-core-foundation",
|
||||
"objc2-foundation",
|
||||
@@ -1891,7 +1943,7 @@ version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d87d638e33c06f577498cbcc50491496a3ed4246998a7fbba7ccb98b1e7eab22"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"block2",
|
||||
"objc2",
|
||||
"objc2-cloud-kit",
|
||||
@@ -1916,6 +1968,21 @@ dependencies = [
|
||||
"objc2-foundation",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ocrs"
|
||||
version = "0.12.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a5379fdd3f11522b5a2ff53017a189463dabf5d0a9c915cb3eb97fabec4ea11c"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"rayon",
|
||||
"rten",
|
||||
"rten-imageproc",
|
||||
"rten-tensor",
|
||||
"thiserror 2.0.18",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "once_cell"
|
||||
version = "1.21.4"
|
||||
@@ -2142,7 +2209,7 @@ version = "0.18.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"crc32fast",
|
||||
"fdeflate",
|
||||
"flate2",
|
||||
@@ -2361,13 +2428,33 @@ dependencies = [
|
||||
"getrandom 0.3.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rayon"
|
||||
version = "1.12.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d"
|
||||
dependencies = [
|
||||
"either",
|
||||
"rayon-core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rayon-core"
|
||||
version = "1.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91"
|
||||
dependencies = [
|
||||
"crossbeam-deque",
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "redox_syscall"
|
||||
version = "0.5.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2376,7 +2463,7 @@ version = "0.7.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4666a1a60d8412eab19d94f6d13dcc9cea0a5ef4fdf6a5db306537413c661b1b"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2496,19 +2583,135 @@ dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "43c230fa4ade87c913f61dbd911b7eb0d49460ceff3f1e4fabc837fac191137c"
|
||||
dependencies = [
|
||||
"flatbuffers",
|
||||
"num_cpus",
|
||||
"rayon",
|
||||
"rten-base",
|
||||
"rten-gemm",
|
||||
"rten-model-file",
|
||||
"rten-onnx",
|
||||
"rten-shape-inference",
|
||||
"rten-simd",
|
||||
"rten-tensor",
|
||||
"rten-vecmath",
|
||||
"rustc-hash",
|
||||
"smallvec",
|
||||
"typeid",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-base"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2738cf8bb4c27f828ac788d01ccf4e367e8e773cfec6851f81851b5211de6a79"
|
||||
dependencies = [
|
||||
"rayon",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-gemm"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "330a81a0ca209fb5ce21bd17efa0bd287d5881c6cebfbff0b21c4294a1a14a9e"
|
||||
dependencies = [
|
||||
"rayon",
|
||||
"rten-base",
|
||||
"rten-simd",
|
||||
"rten-tensor",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-imageproc"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d5f148e7e941fb5727b9046a5fa1b45525543d5105f14b384fd9261df0ee49bc"
|
||||
dependencies = [
|
||||
"rten-tensor",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-model-file"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ed2f8d270f07ab1bbfff47250c6039f6caa5da59d6da7d74f66aa48559aa6fea"
|
||||
dependencies = [
|
||||
"flatbuffers",
|
||||
"rten-base",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-onnx"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "23086eef75bfb55278cb0b45cf9f5a877d466d914914aafebee4ffca9b24d20c"
|
||||
|
||||
[[package]]
|
||||
name = "rten-shape-inference"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8e8a913c7ca40e2bfbb2a0cd447cce56b33ab19435f56693271a2ef37cf58984"
|
||||
dependencies = [
|
||||
"rten-tensor",
|
||||
"smallvec",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-simd"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b19a0032dfcb70dd20960c1c51a37674b237586cbc1ce586f45b46605d108e82"
|
||||
|
||||
[[package]]
|
||||
name = "rten-tensor"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05dc744a270aa32d154f1a3df8e48740ccc1be9dfbcf23295ada66d83aa98de6"
|
||||
dependencies = [
|
||||
"rayon",
|
||||
"rten-base",
|
||||
"smallvec",
|
||||
"typeid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rten-vecmath"
|
||||
version = "0.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9574ddebf5671bc08ceb76e2e1638fadc57fdeff318634eab2c29e9a803cff64"
|
||||
dependencies = [
|
||||
"rten-base",
|
||||
"rten-simd",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustc-hash"
|
||||
version = "2.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe"
|
||||
|
||||
[[package]]
|
||||
name = "rustc_version"
|
||||
version = "0.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92"
|
||||
dependencies = [
|
||||
"semver",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustix"
|
||||
version = "0.38.44"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fdb5bc1ae2baa591800df16c9ca78619bf65c0488b41b96ccec5d11220d8c154"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys 0.4.15",
|
||||
@@ -2521,7 +2724,7 @@ version = "1.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys 0.12.1",
|
||||
@@ -2603,7 +2806,7 @@ version = "0.25.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4eb30575f3638fc8f6815f448d50cb1a2e255b0897985c8c59f4d37b72a07b06"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"cssparser",
|
||||
"derive_more",
|
||||
"fxhash",
|
||||
@@ -2911,7 +3114,7 @@ checksum = "aa003f0038df784eb8fecbbac13affe3da23b45194bd57dba231c8f48199c526"
|
||||
dependencies = [
|
||||
"atoi",
|
||||
"base64",
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"byteorder",
|
||||
"bytes",
|
||||
"chrono",
|
||||
@@ -2955,7 +3158,7 @@ checksum = "db58fcd5a53cf07c184b154801ff91347e4c30d17a3562a635ff028ad5deda46"
|
||||
dependencies = [
|
||||
"atoi",
|
||||
"base64",
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"byteorder",
|
||||
"chrono",
|
||||
"crc",
|
||||
@@ -3317,7 +3520,7 @@ version = "0.6.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "68d6fdd9f81c2819c9a8b0e0cd91660e7746a8e6ea2ba7c6b2b057985f6bcb51"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"bytes",
|
||||
"futures-util",
|
||||
"http",
|
||||
@@ -3428,6 +3631,12 @@ dependencies = [
|
||||
"utf-8",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "typeid"
|
||||
version = "1.0.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bc7d623258602320d5c55d1bc22793b57daff0ec7efc270ea7d55ce1d5f5471c"
|
||||
|
||||
[[package]]
|
||||
name = "typenum"
|
||||
version = "1.20.0"
|
||||
@@ -3668,7 +3877,7 @@ version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"hashbrown 0.15.5",
|
||||
"indexmap",
|
||||
"semver",
|
||||
@@ -4096,7 +4305,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"bitflags",
|
||||
"bitflags 2.11.1",
|
||||
"indexmap",
|
||||
"log",
|
||||
"serde",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "mangalord"
|
||||
version = "0.88.0"
|
||||
version = "0.128.25"
|
||||
edition = "2021"
|
||||
default-run = "mangalord"
|
||||
|
||||
@@ -35,7 +35,6 @@ dotenvy = "0.15"
|
||||
argon2 = "0.5"
|
||||
rand = "0.8"
|
||||
sha2 = "0.10"
|
||||
subtle = "2"
|
||||
base64 = "0.22"
|
||||
# Image decode + downscale for the analysis worker (keep the page image
|
||||
# under the local vision model's token budget). Only the manga page formats.
|
||||
@@ -52,6 +51,8 @@ sysinfo = { version = "0.32", default-features = false, features = ["system", "c
|
||||
nix = { version = "0.29", features = ["fs"] }
|
||||
scraper = "0.20"
|
||||
reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "socks", "cookies", "stream", "json"] }
|
||||
ocrs = "0.12"
|
||||
rten = "0.24"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
@@ -60,6 +61,7 @@ http-body-util = "0.1"
|
||||
mime = "0.3"
|
||||
futures-util = "0.3"
|
||||
tokio = { version = "1", features = ["test-util"] }
|
||||
image = { version = "0.25", default-features = false, features = ["jpeg", "png", "webp"] }
|
||||
|
||||
# Trim debug builds: keep line numbers in panics / backtraces but drop the
|
||||
# full DWARF info (variable-level inspection in gdb/lldb). With a sqlx +
|
||||
|
||||
@@ -58,6 +58,20 @@ WORKDIR /app
|
||||
COPY --from=builder /app/target/release/mangalord /usr/local/bin/mangalord
|
||||
COPY --from=builder /app/migrations /app/migrations
|
||||
|
||||
# OCR models for the default `ANALYSIS_BACKEND=ocr` (ocrs) engine. The two
|
||||
# `.rten` files are pulled at build time into /models, where the runtime's
|
||||
# `OCRS_DETECTION_MODEL` / `OCRS_RECOGNITION_MODEL` defaults point. They're a
|
||||
# few MB each and pure data (no native code), so they bake cleanly into the
|
||||
# image and need no manual setup on the Pi. Set INSTALL_OCR_MODELS=false to
|
||||
# skip (e.g. for a vision-only deploy that never runs ocrs).
|
||||
ARG INSTALL_OCR_MODELS=true
|
||||
ARG OCRS_MODEL_BASE_URL=https://ocrs-models.s3-accelerate.amazonaws.com
|
||||
RUN if [ "$INSTALL_OCR_MODELS" = "true" ]; then \
|
||||
mkdir -p /models \
|
||||
&& curl -fsSL "${OCRS_MODEL_BASE_URL}/text-detection.rten" -o /models/text-detection.rten \
|
||||
&& curl -fsSL "${OCRS_MODEL_BASE_URL}/text-recognition.rten" -o /models/text-recognition.rten; \
|
||||
fi
|
||||
|
||||
ENV STORAGE_DIR=/var/lib/mangalord/storage
|
||||
# Pre-create the storage dir so the entrypoint doesn't need to
|
||||
# mkdir-as-root and so the named volume mount inherits the right
|
||||
|
||||
9
backend/migrations/0034_api_tokens_expires_at.sql
Normal file
9
backend/migrations/0034_api_tokens_expires_at.sql
Normal file
@@ -0,0 +1,9 @@
|
||||
-- Optional expiry for bot API tokens. NULL = never expires (the prior
|
||||
-- behaviour, preserved for all existing rows). When set, `find_active`
|
||||
-- rejects the token past this instant, mirroring the sessions table's
|
||||
-- `expires_at > now()` gate.
|
||||
ALTER TABLE api_tokens ADD COLUMN expires_at TIMESTAMPTZ;
|
||||
|
||||
-- Partial index to keep the active-token lookup cheap once expiries exist.
|
||||
CREATE INDEX api_tokens_expires_at_idx ON api_tokens (expires_at)
|
||||
WHERE expires_at IS NOT NULL;
|
||||
15
backend/migrations/0035_manga_reactions.sql
Normal file
15
backend/migrations/0035_manga_reactions.sql
Normal file
@@ -0,0 +1,15 @@
|
||||
-- Per-user like/dislike reactions on mangas — a private taste signal that
|
||||
-- powers content-based recommendations. One row per (user, manga); the
|
||||
-- `reaction` column toggles between 'like' and 'dislike', and clearing a
|
||||
-- reaction deletes the row. Reactions are never exposed publicly (no counts).
|
||||
CREATE TABLE manga_reactions (
|
||||
user_id uuid NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
manga_id uuid NOT NULL REFERENCES mangas(id) ON DELETE CASCADE,
|
||||
reaction text NOT NULL CHECK (reaction IN ('like', 'dislike')),
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (user_id, manga_id)
|
||||
);
|
||||
|
||||
-- Recommendations aggregate a user's reacted mangas by tag; the PK covers
|
||||
-- per-user lookups, this covers the reverse (all reactions on a manga).
|
||||
CREATE INDEX manga_reactions_manga_idx ON manga_reactions (manga_id);
|
||||
103
backend/migrations/0036_manga_content_warnings.sql
Normal file
103
backend/migrations/0036_manga_content_warnings.sql
Normal file
@@ -0,0 +1,103 @@
|
||||
-- Denormalized manga -> content-warning set.
|
||||
--
|
||||
-- The list filter previously tested each candidate manga with a correlated
|
||||
-- `page_content_warnings -> pages -> chapters` join, i.e. O(mangas * pages) on
|
||||
-- every filtered list AND its count. This table holds the DISTINCT union of a
|
||||
-- manga's page warnings so the filter is a single indexed lookup.
|
||||
--
|
||||
-- Kept in sync by triggers that recompute an affected manga's set from current
|
||||
-- data (the set is tiny — at most the five moderation labels — so a
|
||||
-- delete-and-reinsert per change is cheap and always correct, sidestepping the
|
||||
-- cascade-ordering hazards of incremental maintenance).
|
||||
|
||||
CREATE TABLE manga_content_warnings (
|
||||
manga_id uuid NOT NULL REFERENCES mangas(id) ON DELETE CASCADE,
|
||||
warning text NOT NULL
|
||||
CHECK (warning IN ('sexual', 'nudity', 'gore', 'violence', 'disturbing')),
|
||||
PRIMARY KEY (manga_id, warning)
|
||||
);
|
||||
|
||||
-- warning -> mangas, for the include/exclude list filters.
|
||||
CREATE INDEX manga_content_warnings_warning_idx ON manga_content_warnings (warning);
|
||||
|
||||
-- Recompute a single manga's warning set from the live per-page rows.
|
||||
CREATE OR REPLACE FUNCTION mcw_refresh_for_manga(mid uuid) RETURNS void AS $$
|
||||
BEGIN
|
||||
-- Skip when the manga is gone (e.g. mid-cascade of a manga delete) so we
|
||||
-- never re-insert a row that would violate the FK / resurrect a deleted set.
|
||||
IF NOT EXISTS (SELECT 1 FROM mangas WHERE id = mid) THEN
|
||||
RETURN;
|
||||
END IF;
|
||||
DELETE FROM manga_content_warnings WHERE manga_id = mid;
|
||||
INSERT INTO manga_content_warnings (manga_id, warning)
|
||||
SELECT DISTINCT mid, pw.warning
|
||||
FROM page_content_warnings pw
|
||||
JOIN pages p ON p.id = pw.page_id
|
||||
JOIN chapters c ON c.id = p.chapter_id
|
||||
WHERE c.manga_id = mid;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
-- page_content_warnings changed for a page: refresh that page's manga.
|
||||
CREATE OR REPLACE FUNCTION mcw_on_pcw_change() RETURNS trigger AS $$
|
||||
DECLARE
|
||||
mid uuid;
|
||||
pid uuid := COALESCE(NEW.page_id, OLD.page_id);
|
||||
BEGIN
|
||||
-- The page (and thus chapter) may already be gone when this fires as part
|
||||
-- of a pages/chapters cascade; in that case the pages/chapters triggers do
|
||||
-- the refresh instead, so a missing join here is harmless.
|
||||
SELECT c.manga_id INTO mid
|
||||
FROM pages p JOIN chapters c ON c.id = p.chapter_id
|
||||
WHERE p.id = pid;
|
||||
IF mid IS NOT NULL THEN
|
||||
PERFORM mcw_refresh_for_manga(mid);
|
||||
END IF;
|
||||
RETURN NULL;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
CREATE TRIGGER mcw_pcw_ins AFTER INSERT ON page_content_warnings
|
||||
FOR EACH ROW EXECUTE FUNCTION mcw_on_pcw_change();
|
||||
CREATE TRIGGER mcw_pcw_del AFTER DELETE ON page_content_warnings
|
||||
FOR EACH ROW EXECUTE FUNCTION mcw_on_pcw_change();
|
||||
|
||||
-- A page was deleted (directly, or via a chapter cascade): its
|
||||
-- page_content_warnings rows are already cascade-gone, so recompute from the
|
||||
-- chapter's manga. If the chapter is gone too, the chapters trigger covers it.
|
||||
CREATE OR REPLACE FUNCTION mcw_on_page_delete() RETURNS trigger AS $$
|
||||
DECLARE
|
||||
mid uuid;
|
||||
BEGIN
|
||||
SELECT c.manga_id INTO mid FROM chapters c WHERE c.id = OLD.chapter_id;
|
||||
IF mid IS NOT NULL THEN
|
||||
PERFORM mcw_refresh_for_manga(mid);
|
||||
END IF;
|
||||
RETURN NULL;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
CREATE TRIGGER mcw_page_del AFTER DELETE ON pages
|
||||
FOR EACH ROW EXECUTE FUNCTION mcw_on_page_delete();
|
||||
|
||||
-- A chapter was deleted (directly, or via a manga cascade): recompute from the
|
||||
-- chapter's manga. `chapters.manga_id` is on the row, so it is always available
|
||||
-- even after the child pages have cascaded; the refresh no-ops when the manga
|
||||
-- itself is being deleted.
|
||||
CREATE OR REPLACE FUNCTION mcw_on_chapter_delete() RETURNS trigger AS $$
|
||||
BEGIN
|
||||
PERFORM mcw_refresh_for_manga(OLD.manga_id);
|
||||
RETURN NULL;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
CREATE TRIGGER mcw_chapter_del AFTER DELETE ON chapters
|
||||
FOR EACH ROW EXECUTE FUNCTION mcw_on_chapter_delete();
|
||||
|
||||
-- Backfill from existing per-page rows.
|
||||
INSERT INTO manga_content_warnings (manga_id, warning)
|
||||
SELECT DISTINCT c.manga_id, pw.warning
|
||||
FROM page_content_warnings pw
|
||||
JOIN pages p ON p.id = pw.page_id
|
||||
JOIN chapters c ON c.id = p.chapter_id
|
||||
ON CONFLICT DO NOTHING;
|
||||
51
backend/migrations/0037_mangas_sort_author.sql
Normal file
51
backend/migrations/0037_mangas_sort_author.sql
Normal file
@@ -0,0 +1,51 @@
|
||||
-- Precomputed author sort key for `?sort=author`.
|
||||
--
|
||||
-- The author sort used a correlated `min(lower(a.name))` subquery as the ORDER
|
||||
-- BY key, evaluated per filter-matching row before LIMIT — it scaled worse than
|
||||
-- the indexed date/title sorts. Materialize the same value on `mangas` so the
|
||||
-- sort is a plain indexed column read.
|
||||
--
|
||||
-- `sort_author` = the alphabetically-first attached author's lowercased name,
|
||||
-- or NULL when the manga has no authors (kept last via NULLS LAST in the query).
|
||||
-- Author names are immutable (authors are upserted by unique lowercased name and
|
||||
-- never renamed), so the value only changes when the manga_authors join changes
|
||||
-- — maintained by the triggers below.
|
||||
|
||||
ALTER TABLE mangas ADD COLUMN sort_author text;
|
||||
|
||||
CREATE INDEX mangas_sort_author_idx ON mangas (sort_author, id);
|
||||
|
||||
-- Backfill from existing links.
|
||||
UPDATE mangas m
|
||||
SET sort_author = (
|
||||
SELECT min(lower(a.name))
|
||||
FROM manga_authors ma
|
||||
JOIN authors a ON a.id = ma.author_id
|
||||
WHERE ma.manga_id = m.id
|
||||
);
|
||||
|
||||
CREATE OR REPLACE FUNCTION refresh_manga_sort_author(mid uuid) RETURNS void AS $$
|
||||
BEGIN
|
||||
UPDATE mangas
|
||||
SET sort_author = (
|
||||
SELECT min(lower(a.name))
|
||||
FROM manga_authors ma
|
||||
JOIN authors a ON a.id = ma.author_id
|
||||
WHERE ma.manga_id = mid
|
||||
)
|
||||
WHERE id = mid;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
CREATE OR REPLACE FUNCTION mangas_sort_author_on_ma_change() RETURNS trigger AS $$
|
||||
BEGIN
|
||||
-- On a manga cascade-delete the UPDATE simply no-ops (row already gone).
|
||||
PERFORM refresh_manga_sort_author(COALESCE(NEW.manga_id, OLD.manga_id));
|
||||
RETURN NULL;
|
||||
END;
|
||||
$$ LANGUAGE plpgsql;
|
||||
|
||||
CREATE TRIGGER manga_authors_sort_author_ins AFTER INSERT ON manga_authors
|
||||
FOR EACH ROW EXECUTE FUNCTION mangas_sort_author_on_ma_change();
|
||||
CREATE TRIGGER manga_authors_sort_author_del AFTER DELETE ON manga_authors
|
||||
FOR EACH ROW EXECUTE FUNCTION mangas_sort_author_on_ma_change();
|
||||
44
backend/migrations/0038_genres_name_lower_unique.sql
Normal file
44
backend/migrations/0038_genres_name_lower_unique.sql
Normal file
@@ -0,0 +1,44 @@
|
||||
-- Enforce case-insensitive genre uniqueness, matching `authors` and `tags`
|
||||
-- (both got `UNIQUE (lower(name))` in 0009). `genres` only had a case-SENSITIVE
|
||||
-- `name UNIQUE`, but the crawler's `sync_genres` treats genres as
|
||||
-- case-insensitive (`WHERE lower(name) = lower($1)`) and can INSERT a new row
|
||||
-- from a source string. Under a race (or across ticks) two different-cased
|
||||
-- strings — "Isekai" vs "isekai" — could both miss the pre-check and both
|
||||
-- insert, since the exact-name unique doesn't collide. That leaves duplicate
|
||||
-- genre rows for one logical genre.
|
||||
|
||||
-- Heal any pre-existing case-variant duplicates before adding the index, so
|
||||
-- `sqlx::migrate!` can't crash on a dirty table (mirrors 0031's pre-dedup).
|
||||
-- Canonical = lowest id per lower(name).
|
||||
|
||||
-- 1. Ensure every manga linked to any case-variant also has the canonical link.
|
||||
-- INSERT ... ON CONFLICT DO NOTHING is collision-proof: SELECT DISTINCT
|
||||
-- collapses multiple variants of one manga to a single canonical row, and the
|
||||
-- ON CONFLICT absorbs a canonical link that already exists. (A prior
|
||||
-- UPDATE-repoint could set two non-canonical rows of the SAME manga to the
|
||||
-- same canonical id in a single statement and violate the manga_genres PK,
|
||||
-- rolling the migration back and wedging startup.)
|
||||
INSERT INTO manga_genres (manga_id, genre_id)
|
||||
SELECT DISTINCT mg.manga_id, k.keep_id
|
||||
FROM manga_genres mg
|
||||
JOIN genres g ON mg.genre_id = g.id
|
||||
JOIN (SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id
|
||||
FROM genres GROUP BY lower(name)) k
|
||||
ON lower(g.name) = k.lname
|
||||
WHERE g.id <> k.keep_id
|
||||
ON CONFLICT (manga_id, genre_id) DO NOTHING;
|
||||
|
||||
-- 2. Every non-canonical link now has a canonical sibling — drop the variants.
|
||||
DELETE FROM manga_genres mg
|
||||
USING genres g,
|
||||
(SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id
|
||||
FROM genres GROUP BY lower(name)) k
|
||||
WHERE mg.genre_id = g.id AND lower(g.name) = k.lname AND g.id <> k.keep_id;
|
||||
|
||||
-- Remove the now-orphaned duplicate genre rows.
|
||||
DELETE FROM genres g
|
||||
USING (SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id
|
||||
FROM genres GROUP BY lower(name)) k
|
||||
WHERE lower(g.name) = k.lname AND g.id <> k.keep_id;
|
||||
|
||||
CREATE UNIQUE INDEX genres_name_lower_uniq ON genres (lower(name));
|
||||
18
backend/migrations/0039_crawler_jobs_running_lease_idx.sql
Normal file
18
backend/migrations/0039_crawler_jobs_running_lease_idx.sql
Normal file
@@ -0,0 +1,18 @@
|
||||
-- Index the crashed-worker-recovery arm of the job lease predicate.
|
||||
--
|
||||
-- lease/lease_kinds match:
|
||||
-- WHERE (state = 'pending' OR (state = 'running' AND leased_until < now()))
|
||||
-- crawler_jobs_ready_idx (0016) is `ON (scheduled_at) WHERE state = 'pending'`,
|
||||
-- so it covers only the pending arm. The `state = 'running' AND leased_until`
|
||||
-- arm had no usable index, so Postgres could not BitmapOr the two arms and
|
||||
-- degraded to a sequential scan of the ENTIRE crawler_jobs table — including all
|
||||
-- done/dead rows not yet reaped — on every lease poll, by every worker,
|
||||
-- continuously (audit H2, the headline performance finding).
|
||||
--
|
||||
-- A partial index on leased_until over just the running rows makes the recovery
|
||||
-- arm index-backed. `state = 'running'` is an immutable predicate (now() stays
|
||||
-- in the query, not the index). The running set is tiny (in-flight jobs only),
|
||||
-- so the index is cheap to maintain.
|
||||
CREATE INDEX crawler_jobs_running_lease_idx
|
||||
ON crawler_jobs (leased_until)
|
||||
WHERE state = 'running';
|
||||
@@ -0,0 +1,9 @@
|
||||
-- Index the retention reaper's scan. reap_terminal deletes
|
||||
-- WHERE state IN ('done','dead') AND updated_at < now() - interval
|
||||
-- which had no supporting index, so each daily sweep sequential-scanned the whole
|
||||
-- crawler_jobs table to find expired terminal rows (audit M5). A partial index on
|
||||
-- updated_at over just the terminal states makes the batched reaper's per-batch
|
||||
-- SELECT index-backed.
|
||||
CREATE INDEX crawler_jobs_terminal_reap_idx
|
||||
ON crawler_jobs (updated_at)
|
||||
WHERE state IN ('done', 'dead');
|
||||
@@ -37,6 +37,22 @@ const LEASE_HEARTBEAT: Duration = Duration::from_secs(20);
|
||||
/// long enough not to hammer `/health` while vision is down.
|
||||
const READINESS_POLL: Duration = Duration::from_secs(2);
|
||||
|
||||
/// Longest an idle analysis worker waits between lease polls, bounding the
|
||||
/// exponential [`idle_backoff`].
|
||||
const IDLE_BACKOFF_CAP: Duration = Duration::from_secs(30);
|
||||
|
||||
/// Exponential idle backoff (1s, 2s, 4s … capped at [`IDLE_BACKOFF_CAP`]).
|
||||
/// `consecutive_empty` is the number of empty lease polls so far (0 on the first
|
||||
/// miss); reset to 0 the moment a job is leased. Replaces the old flat 1s sleep
|
||||
/// so an idle worker isn't firing a `SELECT … FOR UPDATE SKIP LOCKED` lease
|
||||
/// query every second (audit: analysis daemon idle poll). Mirrors the crawler
|
||||
/// daemon's backoff.
|
||||
fn idle_backoff(consecutive_empty: u32) -> Duration {
|
||||
let cap = IDLE_BACKOFF_CAP.as_secs();
|
||||
let secs = 1u64.checked_shl(consecutive_empty).unwrap_or(cap).min(cap);
|
||||
Duration::from_secs(secs)
|
||||
}
|
||||
|
||||
/// The unit of work: analyze one page. Implemented by
|
||||
/// [`RealAnalyzeDispatcher`] in production and stubbed in tests.
|
||||
#[async_trait]
|
||||
@@ -135,6 +151,9 @@ impl WorkerContext {
|
||||
// Last observed readiness, so we log only on transitions (not every
|
||||
// poll). `None` until the first probe.
|
||||
let mut was_ready: Option<bool> = None;
|
||||
// Empty lease polls seen in a row, driving the idle backoff. Reset to 0
|
||||
// the moment a job is leased.
|
||||
let mut consecutive_empty: u32 = 0;
|
||||
loop {
|
||||
if self.cancel.is_cancelled() {
|
||||
tracing::info!(worker = self.id, "analysis worker: shutdown");
|
||||
@@ -178,11 +197,15 @@ impl WorkerContext {
|
||||
}
|
||||
};
|
||||
let Some(lease) = leases.into_iter().next() else {
|
||||
if self.sleep_or_cancel(Duration::from_secs(1)).await {
|
||||
let wait = idle_backoff(consecutive_empty);
|
||||
consecutive_empty = consecutive_empty.saturating_add(1);
|
||||
if self.sleep_or_cancel(wait).await {
|
||||
return;
|
||||
}
|
||||
continue;
|
||||
};
|
||||
// Leased work — drop back to a tight poll cadence.
|
||||
consecutive_empty = 0;
|
||||
self.process_lease(lease).await;
|
||||
}
|
||||
}
|
||||
@@ -393,19 +416,20 @@ impl AnalyzeDispatcher for RealAnalyzeDispatcher {
|
||||
// Page was deleted between enqueue and dispatch — nothing to do.
|
||||
return Ok(());
|
||||
};
|
||||
let bytes = self
|
||||
// Stream through the byte cap so an oversized blob is rejected as it's
|
||||
// read rather than after it's fully buffered into memory (mirrors the
|
||||
// OCR dispatcher).
|
||||
let file = self
|
||||
.storage
|
||||
.get(&page.storage_key)
|
||||
.get_stream(&page.storage_key)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("read page image {}: {e}", page.storage_key))?;
|
||||
if bytes.len() > self.max_image_bytes {
|
||||
anyhow::bail!(
|
||||
"page image {} is {} bytes, over the {} cap",
|
||||
page.storage_key,
|
||||
bytes.len(),
|
||||
self.max_image_bytes
|
||||
);
|
||||
}
|
||||
let bytes =
|
||||
crate::crawler::safety::accumulate_capped(file.stream, self.max_image_bytes)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
anyhow::anyhow!("page image {} over the byte cap: {e}", page.storage_key)
|
||||
})?;
|
||||
let analysis = self.vision.analyze(&bytes, &page.content_type).await?;
|
||||
repo::page_analysis::persist_analysis(&self.db, page_id, &analysis, &self.model).await?;
|
||||
Ok(())
|
||||
@@ -509,6 +533,17 @@ mod tests {
|
||||
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
||||
use tokio::net::TcpListener;
|
||||
|
||||
#[test]
|
||||
fn idle_backoff_grows_exponentially_and_caps() {
|
||||
assert_eq!(idle_backoff(0), Duration::from_secs(1));
|
||||
assert_eq!(idle_backoff(1), Duration::from_secs(2));
|
||||
assert_eq!(idle_backoff(2), Duration::from_secs(4));
|
||||
assert_eq!(idle_backoff(4), Duration::from_secs(16));
|
||||
// Caps at IDLE_BACKOFF_CAP (30s) and never overflows on large counts.
|
||||
assert_eq!(idle_backoff(5), IDLE_BACKOFF_CAP);
|
||||
assert_eq!(idle_backoff(100), IDLE_BACKOFF_CAP);
|
||||
}
|
||||
|
||||
/// Bind an ephemeral port that answers every request with `status_line`
|
||||
/// (e.g. `"200 OK"`), and return its `/health` URL. The probe under test
|
||||
/// only inspects the status code, so a zero-length body is enough.
|
||||
|
||||
@@ -9,5 +9,6 @@
|
||||
|
||||
pub mod daemon;
|
||||
pub mod events;
|
||||
pub mod ocr;
|
||||
pub mod prompt;
|
||||
pub mod vision;
|
||||
|
||||
388
backend/src/analysis/ocr.rs
Normal file
388
backend/src/analysis/ocr.rs
Normal file
@@ -0,0 +1,388 @@
|
||||
//! The in-process OCR analysis backend (the `ocrs` engine).
|
||||
//!
|
||||
//! A lightweight alternative to [`crate::analysis::vision`]: instead of a slow
|
||||
//! local LLM, each page is run through `ocrs` — a pure-Rust detect→recognize
|
||||
//! OCR pipeline on the `rten` runtime. It extracts **text only** (no tags,
|
||||
//! scene description or NSFW flags — those stay the vision backend's job), then
|
||||
//! reuses [`repo::page_analysis::persist_analysis`] so the OCR lines land in
|
||||
//! `page_ocr_text` and the weighted `search_doc` tsvector exactly as the vision
|
||||
//! path produces them. That makes the existing text-search surfaces
|
||||
//! (`/v1/me/page-search` and the tag aggregations) work with no further wiring.
|
||||
//!
|
||||
//! The engine is split behind the [`OcrEngine`] trait so the dispatcher is
|
||||
//! unit-testable without shipping the (multi-MB) `.rten` model files: tests use
|
||||
//! [`test_support::StubOcrEngine`], production uses [`OcrsEngine`].
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use async_trait::async_trait;
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::analysis::daemon::AnalyzeDispatcher;
|
||||
use crate::domain::page_analysis::{OcrResult, SafetyFlag, VisionAnalysis};
|
||||
use crate::repo;
|
||||
use crate::storage::Storage;
|
||||
|
||||
/// The `model` label stamped onto `page_analysis` rows written by this backend.
|
||||
pub const OCR_MODEL_LABEL: &str = "ocrs";
|
||||
|
||||
/// Extracts text lines from a decoded-or-encoded page image. The production
|
||||
/// impl ([`OcrsEngine`]) decodes the bytes itself; the trait takes the raw
|
||||
/// stored image bytes so the dispatcher stays engine-agnostic.
|
||||
pub trait OcrEngine: Send + Sync {
|
||||
/// Run OCR over one page image (the bytes as stored, e.g. PNG/JPEG/WebP).
|
||||
/// Returns the recognized text lines in reading order (top→bottom).
|
||||
fn recognize(&self, image: &[u8]) -> anyhow::Result<Vec<String>>;
|
||||
}
|
||||
|
||||
/// Turn the OCR engine's ordered text lines into the [`VisionAnalysis`] shape
|
||||
/// that [`repo::page_analysis::persist_analysis`] consumes. OCR-only: the tag,
|
||||
/// scene and safety fields are left empty/default. The line `kind` is left
|
||||
/// blank — `persist_analysis` maps an empty kind to the neutral mid-weight
|
||||
/// `OcrKind::Narration` bucket (classifying speech/sfx/… is the deferred
|
||||
/// vision backend's job).
|
||||
pub fn lines_to_analysis(lines: Vec<String>) -> VisionAnalysis {
|
||||
let ocr_results = lines
|
||||
.into_iter()
|
||||
.map(|text| OcrResult { text, kind: String::new(), y: None })
|
||||
.collect();
|
||||
VisionAnalysis {
|
||||
ocr_results,
|
||||
tagging_results: Vec::new(),
|
||||
scene_description: String::new(),
|
||||
safety_flag: SafetyFlag::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Production OCR engine: an `ocrs` detect+recognize pipeline with the two
|
||||
/// `.rten` models loaded once at startup. Cheap to share across workers — the
|
||||
/// recognize path borrows `&self`.
|
||||
pub struct OcrsEngine {
|
||||
engine: ocrs::OcrEngine,
|
||||
/// Hard cap on decoded pixel count (decompression-bomb guard). See
|
||||
/// [`decode_rgb8_within`].
|
||||
max_decode_pixels: u64,
|
||||
}
|
||||
|
||||
impl OcrsEngine {
|
||||
/// Load the detection + recognition models from disk and build the engine.
|
||||
/// Fails (at startup) if either model file is missing or unreadable, so a
|
||||
/// misconfigured path is a loud boot error rather than a per-page failure.
|
||||
///
|
||||
/// `max_decode_pixels` bounds the decoded image size (see
|
||||
/// [`decode_rgb8_within`]) — wired from `AnalysisConfig::ocr_max_decode_pixels`.
|
||||
pub fn from_model_paths(
|
||||
detection: &str,
|
||||
recognition: &str,
|
||||
max_decode_pixels: u64,
|
||||
) -> anyhow::Result<Self> {
|
||||
use anyhow::Context;
|
||||
let detection_model = rten::Model::load_file(detection)
|
||||
.with_context(|| format!("load ocrs detection model {detection}"))?;
|
||||
let recognition_model = rten::Model::load_file(recognition)
|
||||
.with_context(|| format!("load ocrs recognition model {recognition}"))?;
|
||||
let engine = ocrs::OcrEngine::new(ocrs::OcrEngineParams {
|
||||
detection_model: Some(detection_model),
|
||||
recognition_model: Some(recognition_model),
|
||||
..Default::default()
|
||||
})
|
||||
.context("construct ocrs engine")?;
|
||||
Ok(Self { engine, max_decode_pixels })
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode encoded image bytes to RGB8 while refusing decompression bombs.
|
||||
///
|
||||
/// `image::load_from_memory` allocates the full decoded buffer up front, so a
|
||||
/// tiny file declaring enormous dimensions (e.g. a 50000×50000 PNG) inflates
|
||||
/// to billions of bytes and OOM-kills the process. We cap the decoder's
|
||||
/// allocation at `max_decode_pixels` worth of RGBA (4 bytes/px headroom over
|
||||
/// the RGB8 result), which the `image` crate checks against the header
|
||||
/// *before* allocating — so an over-size image fails fast instead of dying.
|
||||
fn decode_rgb8_within(image: &[u8], max_decode_pixels: u64) -> anyhow::Result<image::RgbImage> {
|
||||
use anyhow::Context;
|
||||
use std::io::Cursor;
|
||||
|
||||
let mut reader = image::ImageReader::new(Cursor::new(image))
|
||||
.with_guessed_format()
|
||||
.context("guess page image format for OCR")?;
|
||||
let mut limits = image::Limits::default();
|
||||
// 4 bytes/px (RGBA) gives headroom over the eventual RGB8 buffer and any
|
||||
// single intermediate the decoder allocates per pixel.
|
||||
limits.max_alloc = Some(max_decode_pixels.saturating_mul(4));
|
||||
reader.limits(limits);
|
||||
Ok(reader
|
||||
.decode()
|
||||
.context("decode page image for OCR")?
|
||||
.into_rgb8())
|
||||
}
|
||||
|
||||
impl OcrEngine for OcrsEngine {
|
||||
fn recognize(&self, image: &[u8]) -> anyhow::Result<Vec<String>> {
|
||||
// Decode to RGB8 so `ImageSource` gets a known channel layout,
|
||||
// bounding the decoded size against the decompression-bomb cap.
|
||||
let rgb = decode_rgb8_within(image, self.max_decode_pixels)?;
|
||||
let source = ocrs::ImageSource::from_bytes(rgb.as_raw(), rgb.dimensions())
|
||||
.map_err(|e| anyhow::anyhow!("build OCR image source: {e}"))?;
|
||||
let input = self.engine.prepare_input(source)?;
|
||||
// detect words → group into lines → recognize each line. Mirrors
|
||||
// `OcrEngine::get_text`, but keeps the lines as a Vec instead of
|
||||
// joining them, so each becomes its own `page_ocr_text` row.
|
||||
let words = self.engine.detect_words(&input)?;
|
||||
let line_rects = self.engine.find_text_lines(&input, &words);
|
||||
let lines = self
|
||||
.engine
|
||||
.recognize_text(&input, &line_rects)?
|
||||
.into_iter()
|
||||
.filter_map(|line| line.map(|l| l.to_string()))
|
||||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
Ok(lines)
|
||||
}
|
||||
}
|
||||
|
||||
/// Bound on concurrent CPU-bound OCR inferences, given the number of analysis
|
||||
/// workers and the host's available parallelism.
|
||||
///
|
||||
/// Each worker's `dispatch` fires one `spawn_blocking` OCR run, so without a
|
||||
/// bound `ANALYSIS_WORKERS` blocking tasks can run at once. OCR is fully
|
||||
/// CPU-bound, so running more than there are cores just thrashes the scheduler
|
||||
/// (and balloons the blocking pool). Cap at the core count, but never below 1
|
||||
/// and never above the worker count (more permits than workers is pointless).
|
||||
pub fn ocr_concurrency_limit(workers: usize, cores: usize) -> usize {
|
||||
workers.min(cores.max(1)).max(1)
|
||||
}
|
||||
|
||||
/// Run a CPU-bound OCR closure on the blocking pool while holding an
|
||||
/// **owned** permit for the entire duration of the work.
|
||||
///
|
||||
/// The permit is acquired with `acquire_owned` and moved *into* the
|
||||
/// blocking task rather than being held by the caller's future. This
|
||||
/// matters on cancellation: if the dispatcher future is dropped (graceful
|
||||
/// shutdown), a borrowed permit would be released the instant the future
|
||||
/// unwinds — but `spawn_blocking` work is not cancellable and keeps
|
||||
/// running detached, so the concurrency bound (ANALYSIS_WORKERS) would be
|
||||
/// briefly exceeded. Moving the permit into the task ties the slot's
|
||||
/// lifetime to the actual CPU work.
|
||||
async fn run_ocr_blocking<T, F>(
|
||||
permits: Arc<tokio::sync::Semaphore>,
|
||||
f: F,
|
||||
) -> anyhow::Result<T>
|
||||
where
|
||||
F: FnOnce() -> T + Send + 'static,
|
||||
T: Send + 'static,
|
||||
{
|
||||
let permit = permits
|
||||
.acquire_owned()
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("OCR semaphore closed: {e}"))?;
|
||||
tokio::task::spawn_blocking(move || {
|
||||
let _permit = permit; // held until f() returns, even if the caller is cancelled
|
||||
f()
|
||||
})
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("OCR task join error: {e}"))
|
||||
}
|
||||
|
||||
/// Production dispatcher for the OCR backend: load the page, read its image
|
||||
/// from storage, run OCR on the blocking pool, and persist the lines. Mirrors
|
||||
/// [`crate::analysis::daemon::RealAnalyzeDispatcher`] but with no network I/O.
|
||||
pub struct OcrAnalyzeDispatcher {
|
||||
pub db: PgPool,
|
||||
pub storage: Arc<dyn Storage>,
|
||||
pub engine: Arc<dyn OcrEngine>,
|
||||
pub max_image_bytes: usize,
|
||||
/// Caps concurrent CPU-bound OCR inferences across all workers (see
|
||||
/// [`ocr_concurrency_limit`]). Shared via the `Arc<dyn AnalyzeDispatcher>`,
|
||||
/// so one permit pool covers every worker.
|
||||
pub ocr_permits: Arc<tokio::sync::Semaphore>,
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl AnalyzeDispatcher for OcrAnalyzeDispatcher {
|
||||
async fn dispatch(&self, page_id: Uuid) -> anyhow::Result<()> {
|
||||
let Some(page) = repo::page::find_by_id(&self.db, page_id).await? else {
|
||||
// Page was deleted between enqueue and dispatch — nothing to do.
|
||||
return Ok(());
|
||||
};
|
||||
// Stream the blob through the byte cap so the read itself bails once
|
||||
// the running total exceeds `max_image_bytes` — a plain `get()` would
|
||||
// buffer the whole (possibly huge) image into memory before the cap
|
||||
// could reject it.
|
||||
let file = self
|
||||
.storage
|
||||
.get_stream(&page.storage_key)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("read page image {}: {e}", page.storage_key))?;
|
||||
let bytes =
|
||||
crate::crawler::safety::accumulate_capped(file.stream, self.max_image_bytes)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
anyhow::anyhow!("page image {} over the byte cap: {e}", page.storage_key)
|
||||
})?;
|
||||
// OCR inference is CPU-bound and synchronous — keep it off the async
|
||||
// worker's runtime thread, and gate it behind the shared permit pool so
|
||||
// ANALYSIS_WORKERS > cores can't oversubscribe the blocking pool.
|
||||
let engine = Arc::clone(&self.engine);
|
||||
let lines =
|
||||
run_ocr_blocking(Arc::clone(&self.ocr_permits), move || engine.recognize(&bytes))
|
||||
.await??;
|
||||
let analysis = lines_to_analysis(lines);
|
||||
repo::page_analysis::persist_analysis(&self.db, page_id, &analysis, OCR_MODEL_LABEL).await?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Stubs for the OCR dispatcher's integration tests. Public because the tests
|
||||
/// live in the `tests/` dir (a separate crate).
|
||||
pub mod test_support {
|
||||
use super::*;
|
||||
|
||||
/// An [`OcrEngine`] that returns a fixed set of lines regardless of input,
|
||||
/// so the dispatcher's storage→persist path can be tested without models.
|
||||
pub struct StubOcrEngine {
|
||||
pub lines: Vec<String>,
|
||||
}
|
||||
|
||||
impl StubOcrEngine {
|
||||
pub fn new(lines: &[&str]) -> Arc<Self> {
|
||||
Arc::new(Self { lines: lines.iter().map(|s| s.to_string()).collect() })
|
||||
}
|
||||
}
|
||||
|
||||
impl OcrEngine for StubOcrEngine {
|
||||
fn recognize(&self, _image: &[u8]) -> anyhow::Result<Vec<String>> {
|
||||
Ok(self.lines.clone())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[tokio::test]
|
||||
async fn run_ocr_blocking_holds_permit_until_work_completes_despite_cancellation() {
|
||||
use std::sync::mpsc;
|
||||
use tokio::sync::{Notify, Semaphore};
|
||||
|
||||
let permits = Arc::new(Semaphore::new(1));
|
||||
let (release_tx, release_rx) = mpsc::channel::<()>();
|
||||
let started = Arc::new(Notify::new());
|
||||
|
||||
// Model the dispatch path: acquire an owned permit and run blocking
|
||||
// work that we hold open via a channel.
|
||||
let sem = Arc::clone(&permits);
|
||||
let started2 = Arc::clone(&started);
|
||||
let caller = tokio::spawn(async move {
|
||||
run_ocr_blocking(sem, move || {
|
||||
// Signal that the blocking task now holds the permit, then
|
||||
// block until the test releases us.
|
||||
started2.notify_one();
|
||||
release_rx.recv().ok();
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
});
|
||||
|
||||
// Wait until the blocking task is running and owns the permit.
|
||||
started.notified().await;
|
||||
assert_eq!(permits.available_permits(), 0, "permit taken by blocking work");
|
||||
|
||||
// Cancel the caller future (simulates graceful shutdown). The
|
||||
// blocking task is not cancellable and keeps running detached; with
|
||||
// an *owned* permit the slot must stay occupied. A borrowed permit
|
||||
// would have been released here, regressing the concurrency bound.
|
||||
caller.abort();
|
||||
let _ = caller.await;
|
||||
assert_eq!(
|
||||
permits.available_permits(),
|
||||
0,
|
||||
"owned permit must remain held by the still-running blocking task"
|
||||
);
|
||||
|
||||
// Let the blocking work finish; the permit is then returned.
|
||||
release_tx.send(()).unwrap();
|
||||
let permit = tokio::time::timeout(std::time::Duration::from_secs(5), permits.acquire())
|
||||
.await
|
||||
.expect("permit should be released once blocking work completes")
|
||||
.unwrap();
|
||||
drop(permit);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lines_to_analysis_maps_lines_in_order_and_leaves_rest_empty() {
|
||||
let v = lines_to_analysis(vec!["Hello".to_string(), "world!".to_string()]);
|
||||
assert_eq!(v.ocr_results.len(), 2);
|
||||
assert_eq!(v.ocr_results[0].text, "Hello");
|
||||
assert_eq!(v.ocr_results[1].text, "world!");
|
||||
// OCR-only: kind blank (→ Narration at persist), no tags/scene/safety.
|
||||
assert!(v.ocr_results.iter().all(|r| r.kind.is_empty()));
|
||||
assert!(v.tagging_results.is_empty());
|
||||
assert_eq!(v.scene_description, "");
|
||||
assert!(!v.safety_flag.is_nsfw);
|
||||
assert!(v.safety_flag.content_type.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lines_to_analysis_handles_no_text() {
|
||||
let v = lines_to_analysis(Vec::new());
|
||||
assert!(v.ocr_results.is_empty());
|
||||
}
|
||||
|
||||
/// A minimal valid PNG of `w`×`h` (single black pixel scaled via IHDR is
|
||||
/// not valid; instead encode a real tiny image, then we patch the IHDR
|
||||
/// dimensions for the bomb case). For the happy path we just encode a real
|
||||
/// small image.
|
||||
fn encode_png(w: u32, h: u32) -> Vec<u8> {
|
||||
let img = image::RgbImage::new(w, h);
|
||||
let mut buf = std::io::Cursor::new(Vec::new());
|
||||
image::DynamicImage::ImageRgb8(img)
|
||||
.write_to(&mut buf, image::ImageFormat::Png)
|
||||
.unwrap();
|
||||
buf.into_inner()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_rgb8_within_accepts_normal_page() {
|
||||
// A perfectly ordinary page decodes fine under a generous cap.
|
||||
let png = encode_png(64, 96);
|
||||
let rgb = decode_rgb8_within(&png, 100_000_000).unwrap();
|
||||
assert_eq!(rgb.dimensions(), (64, 96));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_rgb8_within_rejects_oversize_image() {
|
||||
// The same image, but the cap is set below its pixel count: the
|
||||
// allocation limit must trip rather than the decode succeeding. This
|
||||
// is the decompression-bomb guard in miniature — a real bomb declares
|
||||
// a huge size in a few bytes; here we shrink the budget instead.
|
||||
let png = encode_png(2000, 2000); // 4 MP
|
||||
let err = decode_rgb8_within(&png, 1_000).unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("decode page image"),
|
||||
"expected a decode error, got: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ocr_concurrency_limit_caps_at_cores() {
|
||||
// More workers than cores → clamp to cores (don't oversubscribe).
|
||||
assert_eq!(ocr_concurrency_limit(8, 4), 4);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ocr_concurrency_limit_caps_at_workers() {
|
||||
// Fewer workers than cores → only `workers` ever run anyway.
|
||||
assert_eq!(ocr_concurrency_limit(2, 16), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ocr_concurrency_limit_never_zero() {
|
||||
// Degenerate inputs still yield at least one permit.
|
||||
assert_eq!(ocr_concurrency_limit(0, 0), 1);
|
||||
assert_eq!(ocr_concurrency_limit(1, 0), 1);
|
||||
}
|
||||
}
|
||||
@@ -55,6 +55,12 @@ struct SliceParams {
|
||||
overlap: f64,
|
||||
tall_threshold: f64,
|
||||
max_slices: usize,
|
||||
/// Hard cap on the **decoded** pixel count, enforced as an allocation
|
||||
/// limit on the decoder itself. `max_pixels` only downscales *after* a
|
||||
/// full decode, so without this a tiny WebP/JPEG/PNG declaring huge
|
||||
/// dimensions would OOM the blocking worker (decompression bomb).
|
||||
/// Shared with the OCR backend's cap (`ANALYSIS_OCR_MAX_DECODE_PIXELS`).
|
||||
max_decode_pixels: u64,
|
||||
}
|
||||
|
||||
/// Result of the (blocking-pool) prep pass: a self-contained set of byte
|
||||
@@ -86,11 +92,30 @@ enum PreparedAnalysis {
|
||||
/// JPEG-encoder regression indistinguishable from "the page was just
|
||||
/// garbage." Emit a `warn` on each fallback so an operator can grep
|
||||
/// for "vision prep fell through" and tell the two apart.
|
||||
/// Decode an encoded page image with a hard allocation cap so a
|
||||
/// decompression bomb — a tiny WebP/JPEG/PNG header declaring enormous
|
||||
/// dimensions — can't OOM the blocking worker before we get a chance to
|
||||
/// downscale. The `image` crate's default reader applies no such bound;
|
||||
/// format-specific self-limits (strongest for PNG) don't cover WebP/JPEG,
|
||||
/// which manga pages commonly use. `4 bytes/px` (RGBA) leaves headroom
|
||||
/// over any single intermediate the decoder allocates per pixel. Mirrors
|
||||
/// `ocr::decode_rgb8_within`.
|
||||
fn decode_within(image: &[u8], max_decode_pixels: u64) -> Option<DynamicImage> {
|
||||
use std::io::Cursor;
|
||||
let mut reader = image::ImageReader::new(Cursor::new(image))
|
||||
.with_guessed_format()
|
||||
.ok()?;
|
||||
let mut limits = image::Limits::default();
|
||||
limits.max_alloc = Some(max_decode_pixels.saturating_mul(4));
|
||||
reader.limits(limits);
|
||||
reader.decode().ok()
|
||||
}
|
||||
|
||||
fn prepare_analysis(image: &[u8], params: SliceParams) -> PreparedAnalysis {
|
||||
let Some(img) = image::load_from_memory(image).ok() else {
|
||||
let Some(img) = decode_within(image, params.max_decode_pixels) else {
|
||||
tracing::warn!(
|
||||
bytes = image.len(),
|
||||
"vision prep fell through to Undecodable: image::load_from_memory failed"
|
||||
"vision prep fell through to Undecodable: decode failed or exceeded pixel cap"
|
||||
);
|
||||
return PreparedAnalysis::Undecodable;
|
||||
};
|
||||
@@ -162,6 +187,7 @@ impl VisionClient {
|
||||
overlap: cfg.slice_overlap,
|
||||
tall_threshold: cfg.tall_aspect_threshold,
|
||||
max_slices: cfg.max_slices,
|
||||
max_decode_pixels: cfg.ocr_max_decode_pixels,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -779,9 +805,40 @@ mod tests {
|
||||
overlap: 0.12,
|
||||
tall_threshold: 1.6,
|
||||
max_slices: 16,
|
||||
max_decode_pixels: 100_000_000,
|
||||
}
|
||||
}
|
||||
|
||||
fn encode_webp(w: u32, h: u32) -> Vec<u8> {
|
||||
use image::{DynamicImage, ImageFormat, RgbImage};
|
||||
let img = DynamicImage::ImageRgb8(RgbImage::new(w, h));
|
||||
let mut buf = std::io::Cursor::new(Vec::new());
|
||||
img.write_to(&mut buf, ImageFormat::WebP).expect("encode webp");
|
||||
buf.into_inner()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_within_rejects_oversize_webp() {
|
||||
// Non-PNG decompression-bomb coverage: a real 2000×2000 WebP (4 MP)
|
||||
// must be refused when the decode cap is set below its pixel count.
|
||||
// The allocation limit trips inside the decoder rather than the
|
||||
// full frame being materialized. Manga pages are commonly WebP/JPEG,
|
||||
// where the format's own self-limits are weaker than PNG's.
|
||||
let webp = encode_webp(2000, 2000);
|
||||
assert!(decode_within(&webp, 1_000).is_none(), "cap below pixels must reject");
|
||||
// A generous cap decodes the same image fine.
|
||||
assert!(decode_within(&webp, 100_000_000).is_some(), "within cap must decode");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_analysis_falls_back_to_undecodable_on_decode_bomb() {
|
||||
// End-to-end: an over-cap image drives the prep pass to the
|
||||
// Undecodable fallback (raw-bytes path) instead of decoding it.
|
||||
let webp = encode_webp(2000, 2000);
|
||||
let p = SliceParams { max_decode_pixels: 1_000, ..params() };
|
||||
assert!(matches!(prepare_analysis(&webp, p), PreparedAnalysis::Undecodable));
|
||||
}
|
||||
|
||||
fn ocr(text: &str, kind: &str) -> OcrResult {
|
||||
OcrResult {
|
||||
text: text.into(),
|
||||
@@ -1136,6 +1193,7 @@ mod tests {
|
||||
overlap: 0.05,
|
||||
tall_threshold: 1.8,
|
||||
max_slices: 6,
|
||||
max_decode_pixels: 100_000_000,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1212,14 +1270,16 @@ mod tests {
|
||||
// VisionClient pointed at a closed local port. Reqwest fails the
|
||||
// POST immediately (connection refused), but the prep work runs
|
||||
// before the POST is even built.
|
||||
let mut cfg = AnalysisConfig::default();
|
||||
cfg.endpoint = "http://127.0.0.1:1/v1/chat/completions".into();
|
||||
cfg.model = "stub".into();
|
||||
cfg.request_timeout = Duration::from_secs(1);
|
||||
// Tune slice geometry to match the test image's shape.
|
||||
cfg.max_pixels = 1_000_000;
|
||||
cfg.min_slice_height = 100;
|
||||
cfg.tall_aspect_threshold = 1.8;
|
||||
let cfg = AnalysisConfig {
|
||||
endpoint: "http://127.0.0.1:1/v1/chat/completions".into(),
|
||||
model: "stub".into(),
|
||||
request_timeout: Duration::from_secs(1),
|
||||
// Tune slice geometry to match the test image's shape.
|
||||
max_pixels: 1_000_000,
|
||||
min_slice_height: 100,
|
||||
tall_aspect_threshold: 1.8,
|
||||
..Default::default()
|
||||
};
|
||||
let http = reqwest::Client::builder()
|
||||
.timeout(cfg.request_timeout)
|
||||
.no_proxy()
|
||||
|
||||
@@ -58,20 +58,20 @@ async fn stream_status(
|
||||
) -> Sse<impl Stream<Item = Result<Event, Infallible>>> {
|
||||
let rx = state.analysis_events.subscribe();
|
||||
let stream = futures_util::stream::unfold(rx, |mut rx| async move {
|
||||
loop {
|
||||
match rx.recv().await {
|
||||
Ok(ev) => {
|
||||
let event = Event::default()
|
||||
.event("analysis")
|
||||
.json_data(&ev)
|
||||
.unwrap_or_else(|_| Event::default().comment("serialize error"));
|
||||
return Some((Ok(event), rx));
|
||||
}
|
||||
Err(RecvError::Lagged(_)) => {
|
||||
return Some((Ok(Event::default().event("lagged").data("")), rx));
|
||||
}
|
||||
Err(RecvError::Closed) => return None,
|
||||
// One recv per unfold step; the stream driver re-enters for the next
|
||||
// event, so no explicit loop is needed here.
|
||||
match rx.recv().await {
|
||||
Ok(ev) => {
|
||||
let event = Event::default()
|
||||
.event("analysis")
|
||||
.json_data(&ev)
|
||||
.unwrap_or_else(|_| Event::default().comment("serialize error"));
|
||||
Some((Ok(event), rx))
|
||||
}
|
||||
Err(RecvError::Lagged(_)) => {
|
||||
Some((Ok(Event::default().event("lagged").data("")), rx))
|
||||
}
|
||||
Err(RecvError::Closed) => None,
|
||||
}
|
||||
});
|
||||
Sse::new(stream).keep_alive(KeepAlive::default())
|
||||
@@ -328,10 +328,21 @@ fn ensure_enabled(state: &AppState) -> AppResult<()> {
|
||||
async fn reenqueue(
|
||||
State(state): State<AppState>,
|
||||
admin: RequireAdmin,
|
||||
body: Option<Json<ReenqueueBody>>,
|
||||
raw: axum::body::Bytes,
|
||||
) -> AppResult<Json<ReenqueueResponse>> {
|
||||
ensure_enabled(&state)?;
|
||||
let body = body.map(|b| b.0).unwrap_or_default();
|
||||
// An ABSENT/empty body means the default "All" scope. A PRESENT body must be
|
||||
// valid JSON — `Option<Json<T>>` would silently collapse a malformed body to
|
||||
// None and run the full-library default the caller never intended, so parse
|
||||
// the raw bytes ourselves and 422 on a real parse error.
|
||||
let body: ReenqueueBody = if raw.iter().all(u8::is_ascii_whitespace) {
|
||||
ReenqueueBody::default()
|
||||
} else {
|
||||
serde_json::from_slice(&raw).map_err(|e| AppError::ValidationFailed {
|
||||
message: "reenqueue body is not valid JSON".into(),
|
||||
details: json!({ "body": e.to_string() }),
|
||||
})?
|
||||
};
|
||||
|
||||
// Resolve the scope. manga_id and chapter_id are mutually exclusive;
|
||||
// an unknown target is a 404 rather than a silent zero-enqueue.
|
||||
@@ -368,21 +379,15 @@ async fn reenqueue(
|
||||
(repo::page_analysis::ReenqueueScope::All, "analysis", None)
|
||||
};
|
||||
|
||||
// Enqueue + audit in one transaction so a failed audit insert rolls the
|
||||
// enqueue back too — the audit trail can't silently miss a re-enqueue that
|
||||
// actually landed. Both are plain DB writes, so they share the tx cleanly
|
||||
// (the live event + response are emitted only after commit).
|
||||
let mut tx = state.db.begin().await?;
|
||||
let enqueued =
|
||||
repo::page_analysis::enqueue_pages(&state.db, scope, body.only_unanalyzed).await?;
|
||||
|
||||
// Push a live event so connected dashboards mark the in-scope pages as
|
||||
// queued. Skip the no-op (nothing actually enqueued).
|
||||
if enqueued > 0 {
|
||||
state.analysis_events.publish(AnalysisEvent::Enqueued {
|
||||
count: enqueued,
|
||||
manga_id: body.manga_id,
|
||||
chapter_id: body.chapter_id,
|
||||
});
|
||||
}
|
||||
|
||||
repo::page_analysis::enqueue_pages(&mut *tx, scope, body.only_unanalyzed).await?;
|
||||
repo::admin_audit::insert(
|
||||
&state.db,
|
||||
&mut *tx,
|
||||
admin.0.id,
|
||||
"analysis_reenqueue",
|
||||
target_type,
|
||||
@@ -395,6 +400,18 @@ async fn reenqueue(
|
||||
}),
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
|
||||
// Push a live event so connected dashboards mark the in-scope pages as
|
||||
// queued. Skip the no-op (nothing actually enqueued). After commit so a
|
||||
// rolled-back enqueue never emits a phantom event.
|
||||
if enqueued > 0 {
|
||||
state.analysis_events.publish(AnalysisEvent::Enqueued {
|
||||
count: enqueued,
|
||||
manga_id: body.manga_id,
|
||||
chapter_id: body.chapter_id,
|
||||
});
|
||||
}
|
||||
|
||||
Ok(Json(ReenqueueResponse { enqueued }))
|
||||
}
|
||||
@@ -421,8 +438,13 @@ async fn analyze_page(
|
||||
// when a `force=false` job was already pending — the worker would
|
||||
// then pick up the non-force row, hit skip-if-done, ack done, and
|
||||
// the admin saw "queued for re-analysis" with no re-analysis.
|
||||
//
|
||||
// Enqueue + audit in one transaction (via the `_conn` form) so a failed
|
||||
// audit insert rolls the enqueue/upgrade back too — the audit trail can't
|
||||
// miss a force-reanalyze that landed.
|
||||
let mut tx = state.db.begin().await?;
|
||||
let outcome =
|
||||
repo::page_analysis::enqueue_for_page(&state.db, page_id, true).await?;
|
||||
repo::page_analysis::enqueue_for_page_conn(&mut tx, page_id, true).await?;
|
||||
|
||||
let outcome_label = match outcome {
|
||||
repo::page_analysis::EnqueueForPageOutcome::Inserted => "inserted",
|
||||
@@ -430,7 +452,7 @@ async fn analyze_page(
|
||||
repo::page_analysis::EnqueueForPageOutcome::AlreadyEnqueued => "already_force_enqueued",
|
||||
};
|
||||
repo::admin_audit::insert(
|
||||
&state.db,
|
||||
&mut *tx,
|
||||
admin.0.id,
|
||||
"analysis_force_page",
|
||||
"page",
|
||||
@@ -438,6 +460,7 @@ async fn analyze_page(
|
||||
json!({ "force": true, "outcome": outcome_label }),
|
||||
)
|
||||
.await?;
|
||||
tx.commit().await?;
|
||||
|
||||
Ok(Json(AnalyzePageResponse { enqueued: true }))
|
||||
}
|
||||
|
||||
@@ -205,6 +205,13 @@ async fn backfill(
|
||||
resp.more_remaining = more_pages || more_covers;
|
||||
}
|
||||
|
||||
// Audit is written after the action here *by necessity*, not oversight
|
||||
// (unlike reenqueue/analyze_page, which wrap their single DB mutation +
|
||||
// audit in one tx). The backfill is a budgeted scan that interleaves
|
||||
// filesystem `storage.size()` reads with per-batch DB writes across many
|
||||
// iterations; wrapping it in a transaction would hold one open across all
|
||||
// that I/O (a long-running tx). The batch size-writes are also idempotent
|
||||
// recomputations, so a lost audit row has negligible impact.
|
||||
repo::admin_audit::insert(
|
||||
&state.db,
|
||||
admin.0.id,
|
||||
|
||||
@@ -16,7 +16,7 @@ use crate::api::auth::{validate_password, validate_username};
|
||||
use crate::api::pagination::PagedResponse;
|
||||
use crate::app::AppState;
|
||||
use crate::auth::extractor::RequireAdmin;
|
||||
use crate::auth::password::hash_password;
|
||||
use crate::auth::password::hash_password_async;
|
||||
use crate::domain::User;
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::repo;
|
||||
@@ -115,7 +115,7 @@ async fn create_user(
|
||||
// reject (and vice versa).
|
||||
validate_username(username)?;
|
||||
validate_password(&input.password)?;
|
||||
let pwhash = hash_password(&input.password)?;
|
||||
let pwhash = hash_password_async(input.password.clone()).await?;
|
||||
let user = repo::user::admin_create_user(
|
||||
&state.db,
|
||||
actor.id,
|
||||
|
||||
@@ -17,8 +17,8 @@ use serde::{Deserialize, Serialize};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::app::AppState;
|
||||
use crate::auth::extractor::{CurrentUser, SESSION_COOKIE_NAME};
|
||||
use crate::auth::password::{hash_password, verify_password};
|
||||
use crate::auth::extractor::{ClientIp, CurrentUser, SESSION_COOKIE_NAME};
|
||||
use crate::auth::password::{hash_password_async, verify_password_async};
|
||||
use crate::auth::token::{generate_token, hash_token};
|
||||
use crate::config::AuthConfig;
|
||||
use crate::domain::user_preferences::{READER_GAPS, READER_MODES};
|
||||
@@ -38,7 +38,7 @@ pub fn routes() -> Router<AppState> {
|
||||
"/auth/me/preferences",
|
||||
get(get_preferences).patch(update_preferences),
|
||||
)
|
||||
.route("/auth/tokens", post(create_token))
|
||||
.route("/auth/tokens", get(list_tokens).post(create_token))
|
||||
.route("/auth/tokens/:id", delete(delete_token))
|
||||
}
|
||||
|
||||
@@ -75,8 +75,16 @@ pub struct AuthResponse {
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct CreateTokenInput {
|
||||
pub name: String,
|
||||
/// Optional lifetime in days. Omit (or `null`) for a non-expiring
|
||||
/// token (the historical behaviour). When set, must be 1..=3650.
|
||||
#[serde(default)]
|
||||
pub expires_in_days: Option<i64>,
|
||||
}
|
||||
|
||||
/// Upper bound on a requested token lifetime (~10 years). A token that needs
|
||||
/// to outlive this should be rotated, not minted once forever.
|
||||
const MAX_TOKEN_EXPIRY_DAYS: i64 = 3650;
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChangePassword {
|
||||
pub current_password: String,
|
||||
@@ -99,6 +107,7 @@ pub struct CreatedTokenResponse {
|
||||
|
||||
async fn register(
|
||||
State(state): State<AppState>,
|
||||
ClientIp(client_ip): ClientIp,
|
||||
jar: CookieJar,
|
||||
Json(input): Json<Credentials>,
|
||||
) -> AppResult<impl IntoResponse> {
|
||||
@@ -106,7 +115,7 @@ async fn register(
|
||||
// the toggle can't be probed for the toggle state via timing —
|
||||
// disabled and enabled paths both consume a token, and disabled
|
||||
// returns 403 instead of running argon2.
|
||||
check_auth_rate_limit(&state, "register")?;
|
||||
check_auth_rate_limit(&state, "register", client_ip)?;
|
||||
// Private mode force-blocks self-registration regardless of
|
||||
// ALLOW_SELF_REGISTER — operators of locked-down instances mint
|
||||
// accounts via `POST /admin/users` instead.
|
||||
@@ -117,7 +126,7 @@ async fn register(
|
||||
validate_username(username)?;
|
||||
validate_password(&input.password)?;
|
||||
|
||||
let pwhash = hash_password(&input.password)?;
|
||||
let pwhash = hash_password_async(input.password.clone()).await?;
|
||||
let user = repo::user::create(&state.db, username, &pwhash).await?;
|
||||
let jar = start_session(&state, &user, jar).await?;
|
||||
Ok((StatusCode::CREATED, jar, Json(AuthResponse { user })))
|
||||
@@ -125,16 +134,21 @@ async fn register(
|
||||
|
||||
async fn login(
|
||||
State(state): State<AppState>,
|
||||
ClientIp(client_ip): ClientIp,
|
||||
jar: CookieJar,
|
||||
Json(input): Json<Credentials>,
|
||||
) -> AppResult<impl IntoResponse> {
|
||||
check_auth_rate_limit(&state, "login")?;
|
||||
check_auth_rate_limit(&state, "login", client_ip)?;
|
||||
let username = input.username.trim();
|
||||
if username.is_empty() || input.password.is_empty() {
|
||||
return Err(AppError::InvalidInput(
|
||||
"username and password are required".into(),
|
||||
));
|
||||
}
|
||||
// Bound the password before argon2 runs — guards BOTH the real-verify and
|
||||
// the dummy-hash timing-equaliser branch below, so a giant password can't
|
||||
// make every login attempt a CPU-DoS.
|
||||
reject_oversized_password(&input.password)?;
|
||||
|
||||
let user = repo::user::find_by_username(&state.db, username).await?;
|
||||
let Some(user) = user else {
|
||||
@@ -142,10 +156,11 @@ async fn login(
|
||||
// response time matches the wrong-password branch — otherwise
|
||||
// an attacker can enumerate usernames by timing the no-user
|
||||
// 401 against the wrong-password 401.
|
||||
let _ = verify_password(&input.password, dummy_password_hash());
|
||||
let _ = verify_password_async(input.password.clone(), dummy_password_hash().to_string())
|
||||
.await;
|
||||
return Err(AppError::Unauthenticated);
|
||||
};
|
||||
if !verify_password(&input.password, &user.password_hash) {
|
||||
if !verify_password_async(input.password.clone(), user.password_hash.clone()).await {
|
||||
return Err(AppError::Unauthenticated);
|
||||
}
|
||||
|
||||
@@ -201,16 +216,20 @@ async fn me(CurrentUser(user): CurrentUser) -> AppResult<Json<AuthResponse>> {
|
||||
async fn change_password(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
ClientIp(client_ip): ClientIp,
|
||||
jar: CookieJar,
|
||||
Json(input): Json<ChangePassword>,
|
||||
) -> AppResult<impl IntoResponse> {
|
||||
check_auth_rate_limit(&state, "change_password")?;
|
||||
if !verify_password(&input.current_password, &user.password_hash) {
|
||||
check_auth_rate_limit(&state, "change_password", client_ip)?;
|
||||
// Cap current_password before verify_password runs argon2 (same DoS
|
||||
// vector as login). new_password is bounded by validate_password below.
|
||||
reject_oversized_password(&input.current_password)?;
|
||||
if !verify_password_async(input.current_password.clone(), user.password_hash.clone()).await {
|
||||
return Err(AppError::Unauthenticated);
|
||||
}
|
||||
validate_password(&input.new_password)?;
|
||||
|
||||
let new_hash = hash_password(&input.new_password)?;
|
||||
let new_hash = hash_password_async(input.new_password.clone()).await?;
|
||||
|
||||
let mut tx = state.db.begin().await?;
|
||||
sqlx::query("UPDATE users SET password_hash = $1 WHERE id = $2")
|
||||
@@ -280,6 +299,22 @@ async fn update_preferences(
|
||||
Ok(Json(saved))
|
||||
}
|
||||
|
||||
/// `GET /auth/tokens` — the caller's bot tokens (newest first). The raw bearer
|
||||
/// is only ever shown once at creation, so this list carries just the metadata
|
||||
/// (name, created/last-used, expiry); `token_hash` is `#[serde(skip)]`.
|
||||
#[derive(Debug, Serialize)]
|
||||
struct TokenListResponse {
|
||||
items: Vec<ApiToken>,
|
||||
}
|
||||
|
||||
async fn list_tokens(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
) -> AppResult<Json<TokenListResponse>> {
|
||||
let items = repo::api_token::list_for_user(&state.db, user.id).await?;
|
||||
Ok(Json(TokenListResponse { items }))
|
||||
}
|
||||
|
||||
async fn create_token(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
@@ -305,8 +340,22 @@ async fn create_token(
|
||||
details: serde_json::json!({ "name": "max 64 characters" }),
|
||||
});
|
||||
}
|
||||
let expires_at = match input.expires_in_days {
|
||||
None => None,
|
||||
Some(days) if (1..=MAX_TOKEN_EXPIRY_DAYS).contains(&days) => {
|
||||
Some(Utc::now() + Duration::days(days))
|
||||
}
|
||||
Some(_) => {
|
||||
return Err(AppError::ValidationFailed {
|
||||
message: "token expiry out of range".into(),
|
||||
details: serde_json::json!({
|
||||
"expires_in_days": format!("must be between 1 and {MAX_TOKEN_EXPIRY_DAYS}")
|
||||
}),
|
||||
});
|
||||
}
|
||||
};
|
||||
let (raw, hash) = generate_token();
|
||||
let token = repo::api_token::create(&state.db, user.id, name, &hash).await?;
|
||||
let token = repo::api_token::create(&state.db, user.id, name, &hash, expires_at).await?;
|
||||
Ok((
|
||||
StatusCode::CREATED,
|
||||
Json(CreatedTokenResponse { token, bearer: raw }),
|
||||
@@ -387,9 +436,13 @@ fn build_expired_cookie(cfg: &AuthConfig) -> Cookie<'static> {
|
||||
/// any one of them in a tight loop should trip the limit. `endpoint`
|
||||
/// is included in the rate-limit-hit log line so operators can tell
|
||||
/// which endpoint is being probed.
|
||||
fn check_auth_rate_limit(state: &AppState, endpoint: &'static str) -> AppResult<()> {
|
||||
fn check_auth_rate_limit(
|
||||
state: &AppState,
|
||||
endpoint: &'static str,
|
||||
client_ip: Option<std::net::IpAddr>,
|
||||
) -> AppResult<()> {
|
||||
use crate::auth::rate_limit::AcquireResult;
|
||||
match state.auth_limiter.try_acquire() {
|
||||
match state.auth_limiter.try_acquire(client_ip) {
|
||||
AcquireResult::Allowed => Ok(()),
|
||||
AcquireResult::Denied { retry_after_secs } => {
|
||||
tracing::warn!(
|
||||
@@ -425,11 +478,67 @@ pub(crate) fn validate_username(u: &str) -> AppResult<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Upper bound on password length (bytes). argon2 has no inherent length
|
||||
/// limit, so without a cap an attacker could submit a multi-megabyte password
|
||||
/// and make every hash a CPU-heavy DoS. 1024 bytes is far longer than any real
|
||||
/// passphrase. Measured in bytes (like the min check) since that's what argon2
|
||||
/// actually processes.
|
||||
pub(crate) const MAX_PASSWORD_BYTES: usize = 1024;
|
||||
|
||||
/// Reject an over-cap password *before* any argon2 work runs. The
|
||||
/// verification paths (login, change-password) don't go through
|
||||
/// [`validate_password`] — they hash/verify the raw input — so without this
|
||||
/// an attacker could submit a multi-megabyte password and turn every
|
||||
/// login/verify into a CPU-DoS. Length-only and independent of whether the
|
||||
/// account exists, so it leaks nothing (no username enumeration).
|
||||
pub(crate) fn reject_oversized_password(p: &str) -> AppResult<()> {
|
||||
if p.len() > MAX_PASSWORD_BYTES {
|
||||
return Err(AppError::InvalidInput(format!(
|
||||
"password must be at most {MAX_PASSWORD_BYTES} bytes"
|
||||
)));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub(crate) fn validate_password(p: &str) -> AppResult<()> {
|
||||
if p.len() < 8 {
|
||||
return Err(AppError::InvalidInput(
|
||||
"password must be at least 8 characters".into(),
|
||||
));
|
||||
}
|
||||
if p.len() > MAX_PASSWORD_BYTES {
|
||||
return Err(AppError::InvalidInput(
|
||||
format!("password must be at most {MAX_PASSWORD_BYTES} bytes"),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn validate_password_rejects_too_short() {
|
||||
assert!(validate_password("short").is_err());
|
||||
assert!(validate_password("1234567").is_err()); // 7 chars
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_password_accepts_in_range() {
|
||||
assert!(validate_password("hunter2hunter2").is_ok());
|
||||
// Exactly at the cap is allowed.
|
||||
assert!(validate_password(&"a".repeat(MAX_PASSWORD_BYTES)).is_ok());
|
||||
// Minimum length boundary.
|
||||
assert!(validate_password("12345678").is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_password_rejects_over_cap() {
|
||||
// One byte past the cap is refused so a giant password can't turn each
|
||||
// login/register into an argon2 CPU-DoS.
|
||||
let too_long = "a".repeat(MAX_PASSWORD_BYTES + 1);
|
||||
let err = validate_password(&too_long).unwrap_err();
|
||||
assert!(matches!(err, AppError::InvalidInput(_)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -72,7 +72,7 @@ async fn create(
|
||||
}
|
||||
}
|
||||
|
||||
let bookmark = repo::bookmark::create(
|
||||
let (bookmark, created) = repo::bookmark::create(
|
||||
&state.db,
|
||||
user.id,
|
||||
input.manga_id,
|
||||
@@ -103,7 +103,13 @@ async fn create(
|
||||
}
|
||||
});
|
||||
|
||||
Ok((StatusCode::CREATED, Json(bookmark)))
|
||||
// 201 for a fresh bookmark, 200 when it already existed (idempotent add).
|
||||
let status = if created {
|
||||
StatusCode::CREATED
|
||||
} else {
|
||||
StatusCode::OK
|
||||
};
|
||||
Ok((status, Json(bookmark)))
|
||||
}
|
||||
|
||||
async fn delete_one(
|
||||
|
||||
@@ -13,7 +13,7 @@ use serde::Deserialize;
|
||||
use serde_json::json;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::api::mangas::{next_field, read_field_bytes};
|
||||
use crate::api::mangas::next_field;
|
||||
use crate::api::pagination::PagedResponse;
|
||||
use crate::app::AppState;
|
||||
use crate::auth::extractor::CurrentUser;
|
||||
@@ -21,7 +21,8 @@ use crate::domain::chapter::NewChapter;
|
||||
use crate::domain::{Chapter, Page};
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::repo;
|
||||
use crate::upload::{parse_image, UploadedImage};
|
||||
use crate::storage::Storage;
|
||||
use crate::upload::{stage_image_part, StagedImage};
|
||||
|
||||
pub fn routes() -> Router<AppState> {
|
||||
Router::new()
|
||||
@@ -69,6 +70,21 @@ async fn get_one(
|
||||
Ok(Json(chapter))
|
||||
}
|
||||
|
||||
/// Add a chapter to a manga.
|
||||
///
|
||||
/// **Authorization is intentionally open**: any authenticated principal —
|
||||
/// a browser session *or* a bot API token — may add a chapter to *any*
|
||||
/// manga, including crawler-imported rows. This is by design: chapters are
|
||||
/// treated as community contributions, unlike the manga record itself
|
||||
/// (title/cover/metadata), whose edits gate through `require_can_edit`
|
||||
/// (see [`crate::api::mangas`]). The `CurrentUser` binding still requires a
|
||||
/// valid identity, so contributions are attributable, just not owner-scoped.
|
||||
///
|
||||
/// This contract is locked by `tests/api_chapters.rs`
|
||||
/// (`non_owner_can_upload_chapter`); changing it to owner-only is a
|
||||
/// deliberate decision, not a drive-by tightening. Until a richer
|
||||
/// contributor/moderation model lands this is acknowledged, intended
|
||||
/// behaviour — see the auth note in CLAUDE.md.
|
||||
async fn create(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
@@ -77,97 +93,178 @@ async fn create(
|
||||
) -> AppResult<(StatusCode, Json<Chapter>)> {
|
||||
repo::manga::get(&state.db, manga_id).await?;
|
||||
|
||||
// Each `page` part is streamed straight to a staging key as it arrives,
|
||||
// so at most one page's bytes sit in memory — the whole chapter is never
|
||||
// buffered (previously every page was held at once, bounded only by the
|
||||
// 200 MiB body limit and amplified by concurrency). The staged blobs are
|
||||
// promoted to their final chapter-scoped keys in `finalize_chapter` once
|
||||
// the chapter id exists; any early exit cleans them up.
|
||||
let upload_id = Uuid::new_v4();
|
||||
let mut metadata: Option<NewChapter> = None;
|
||||
let mut pages: Vec<UploadedImage> = Vec::new();
|
||||
let mut staged: Vec<StagedImage> = Vec::new();
|
||||
|
||||
while let Some(field) = next_field(&mut multipart).await? {
|
||||
match field.name() {
|
||||
Some("metadata") => {
|
||||
let bytes = read_field_bytes(field).await?;
|
||||
metadata =
|
||||
Some(serde_json::from_slice(&bytes).map_err(|e| {
|
||||
let stage_result: AppResult<NewChapter> = async {
|
||||
while let Some(field) = next_field(&mut multipart).await? {
|
||||
match field.name() {
|
||||
Some("metadata") => {
|
||||
let bytes = crate::upload::read_capped(
|
||||
field,
|
||||
crate::upload::MAX_METADATA_BYTES,
|
||||
"metadata",
|
||||
)
|
||||
.await?;
|
||||
metadata = Some(serde_json::from_slice(&bytes).map_err(|e| {
|
||||
AppError::ValidationFailed {
|
||||
message: "metadata is not valid JSON".into(),
|
||||
details: json!({ "metadata": e.to_string() }),
|
||||
}
|
||||
})?);
|
||||
}
|
||||
Some("page") => {
|
||||
if state.upload.max_pages_per_chapter != 0
|
||||
&& staged.len() >= state.upload.max_pages_per_chapter
|
||||
{
|
||||
return Err(AppError::PayloadTooLarge(format!(
|
||||
"chapter exceeds the {}-page limit",
|
||||
state.upload.max_pages_per_chapter
|
||||
)));
|
||||
}
|
||||
let field_name = format!("page[{}]", staged.len());
|
||||
let img = stage_image_part(
|
||||
state.storage.as_ref(),
|
||||
field,
|
||||
upload_id,
|
||||
staged.len(),
|
||||
state.upload.max_file_bytes,
|
||||
&field_name,
|
||||
)
|
||||
.await?;
|
||||
staged.push(img);
|
||||
}
|
||||
_ => continue,
|
||||
}
|
||||
Some("page") => {
|
||||
let bytes = read_field_bytes(field).await?.to_vec();
|
||||
let field_name = format!("page[{}]", pages.len());
|
||||
pages.push(parse_image(bytes, state.upload.max_file_bytes, &field_name)?);
|
||||
}
|
||||
_ => continue,
|
||||
}
|
||||
}
|
||||
|
||||
let metadata = metadata.ok_or_else(|| AppError::ValidationFailed {
|
||||
message: "metadata part is required".into(),
|
||||
details: json!({ "metadata": "required" }),
|
||||
})?;
|
||||
// Chapter number is 1-indexed everywhere (URLs, upload form,
|
||||
// reader). Reject 0 / negative numbers up front so the row never
|
||||
// makes it into the DB. Mirrors the page>=1 rule on bookmarks.
|
||||
if metadata.number < 1 {
|
||||
return Err(AppError::ValidationFailed {
|
||||
message: "chapter number must be 1 or greater".into(),
|
||||
details: json!({ "number": "must be >= 1" }),
|
||||
});
|
||||
}
|
||||
if pages.is_empty() {
|
||||
return Err(AppError::ValidationFailed {
|
||||
message: "at least one page is required".into(),
|
||||
details: json!({ "page": "at least one required" }),
|
||||
});
|
||||
let metadata = metadata.take().ok_or_else(|| AppError::ValidationFailed {
|
||||
message: "metadata part is required".into(),
|
||||
details: json!({ "metadata": "required" }),
|
||||
})?;
|
||||
// Chapter number is 1-indexed everywhere (URLs, upload form,
|
||||
// reader). Reject 0 / negative numbers up front so the row never
|
||||
// makes it into the DB. Mirrors the page>=1 rule on bookmarks.
|
||||
if metadata.number < 1 {
|
||||
return Err(AppError::ValidationFailed {
|
||||
message: "chapter number must be 1 or greater".into(),
|
||||
details: json!({ "number": "must be >= 1" }),
|
||||
});
|
||||
}
|
||||
if staged.is_empty() {
|
||||
return Err(AppError::ValidationFailed {
|
||||
message: "at least one page is required".into(),
|
||||
details: json!({ "page": "at least one required" }),
|
||||
});
|
||||
}
|
||||
Ok(metadata)
|
||||
}
|
||||
.await;
|
||||
|
||||
// Transactional create. If any storage put or page-row insert
|
||||
// fails mid-loop, the chapter row + any earlier page rows are
|
||||
// rolled back so we don't leave a chapter with stale page_count=0
|
||||
// and orphaned page rows. Bytes already written to storage on a
|
||||
// rolled-back transaction become orphans on disk; a future reaper
|
||||
// can sweep them. DB consistency wins over storage tidiness here.
|
||||
let mut tx = state.db.begin().await?;
|
||||
let mut chapter = repo::chapter::create(
|
||||
let metadata = match stage_result {
|
||||
Ok(m) => m,
|
||||
Err(e) => {
|
||||
// Reject before any DB write — remove every page we staged.
|
||||
cleanup_staging(state.storage.as_ref(), &staged).await;
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
|
||||
finalize_chapter(&state, manga_id, user.id, &metadata, &staged).await
|
||||
}
|
||||
|
||||
/// Promote the staged pages into a real chapter, all-or-nothing. Creates the
|
||||
/// chapter row, renames each staged blob to its final chapter-scoped key,
|
||||
/// and inserts the page rows in one transaction. On any failure the DB rolls
|
||||
/// back and every blob (already-finalized and still-staged) is removed, so a
|
||||
/// rejected upload leaves neither partial rows nor orphaned files.
|
||||
async fn finalize_chapter(
|
||||
state: &AppState,
|
||||
manga_id: Uuid,
|
||||
user_id: Uuid,
|
||||
metadata: &NewChapter,
|
||||
staged: &[StagedImage],
|
||||
) -> AppResult<(StatusCode, Json<Chapter>)> {
|
||||
let storage = state.storage.as_ref();
|
||||
let mut tx = match state.db.begin().await {
|
||||
Ok(tx) => tx,
|
||||
Err(e) => {
|
||||
cleanup_staging(storage, staged).await;
|
||||
return Err(e.into());
|
||||
}
|
||||
};
|
||||
let mut chapter = match repo::chapter::create(
|
||||
&mut *tx,
|
||||
manga_id,
|
||||
metadata.number,
|
||||
metadata.title.as_deref(),
|
||||
Some(user.id),
|
||||
Some(user_id),
|
||||
)
|
||||
.await?;
|
||||
.await
|
||||
{
|
||||
Ok(c) => c,
|
||||
Err(e) => {
|
||||
cleanup_staging(storage, staged).await;
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
|
||||
let mut page_ids: Vec<Uuid> = Vec::with_capacity(pages.len());
|
||||
for (idx, page) in pages.iter().enumerate() {
|
||||
let mut page_ids: Vec<Uuid> = Vec::with_capacity(staged.len());
|
||||
let mut finalized: Vec<String> = Vec::with_capacity(staged.len());
|
||||
for (idx, page) in staged.iter().enumerate() {
|
||||
let page_number = (idx + 1) as i32;
|
||||
let nnnn = format!("{:04}", page_number);
|
||||
let key = format!(
|
||||
"mangas/{}/chapters/{}/pages/{}.{}",
|
||||
manga_id, chapter.id, nnnn, page.ext
|
||||
let final_key = format!(
|
||||
"mangas/{}/chapters/{}/pages/{:04}.{}",
|
||||
manga_id, chapter.id, page_number, page.ext
|
||||
);
|
||||
state.storage.put(&key, &page.bytes).await?;
|
||||
let created = repo::page::create(
|
||||
if let Err(e) = storage.rename(&page.staging_key, &final_key).await {
|
||||
cleanup_keys(storage, &finalized).await;
|
||||
cleanup_staging(storage, &staged[idx..]).await;
|
||||
return Err(e.into());
|
||||
}
|
||||
finalized.push(final_key.clone());
|
||||
match repo::page::create(
|
||||
&mut *tx,
|
||||
chapter.id,
|
||||
page_number,
|
||||
&key,
|
||||
&final_key,
|
||||
page.mime,
|
||||
page.bytes.len() as i64,
|
||||
page.size_bytes,
|
||||
)
|
||||
.await?;
|
||||
page_ids.push(created.id);
|
||||
.await
|
||||
{
|
||||
Ok(created) => page_ids.push(created.id),
|
||||
Err(e) => {
|
||||
cleanup_keys(storage, &finalized).await;
|
||||
cleanup_staging(storage, &staged[idx + 1..]).await;
|
||||
return Err(e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let page_count = pages.len() as i32;
|
||||
repo::chapter::set_page_count(&mut *tx, chapter.id, page_count).await?;
|
||||
let page_count = staged.len() as i32;
|
||||
if let Err(e) = repo::chapter::set_page_count(&mut *tx, chapter.id, page_count).await {
|
||||
cleanup_keys(storage, &finalized).await;
|
||||
return Err(e);
|
||||
}
|
||||
chapter.page_count = page_count;
|
||||
// `repo::chapter::create` returned the row before any pages existed, so
|
||||
// its `size_bytes` is a stale 0. Every uploaded page's size was just
|
||||
// captured, so the true total is the sum of their byte lengths — set it
|
||||
// on the response so the 201 body matches the persisted state.
|
||||
chapter.size_bytes = Some(pages.iter().map(|p| p.bytes.len() as i64).sum());
|
||||
// its `size_bytes` is a stale 0. Each staged page carried its byte
|
||||
// length, so their sum is the chapter's true storage — set it on the
|
||||
// response so the 201 body matches the persisted state.
|
||||
chapter.size_bytes = Some(staged.iter().map(|p| p.size_bytes).sum());
|
||||
|
||||
tx.commit().await?;
|
||||
if let Err(e) = tx.commit().await {
|
||||
cleanup_keys(storage, &finalized).await;
|
||||
return Err(e.into());
|
||||
}
|
||||
|
||||
// Enqueue AI content-analysis for each new page. Done after commit so a
|
||||
// rolled-back upload never leaves jobs pointing at nonexistent pages; a
|
||||
@@ -175,9 +272,7 @@ async fn create(
|
||||
// re-enqueue endpoint can backfill).
|
||||
if state.analysis_enabled() {
|
||||
for page_id in page_ids {
|
||||
if let Err(e) =
|
||||
repo::page_analysis::enqueue_for_page(&state.db, page_id, false).await
|
||||
{
|
||||
if let Err(e) = repo::page_analysis::enqueue_for_page(&state.db, page_id, false).await {
|
||||
tracing::warn!(%page_id, error = %e, "failed to enqueue page analysis");
|
||||
}
|
||||
}
|
||||
@@ -186,6 +281,21 @@ async fn create(
|
||||
Ok((StatusCode::CREATED, Json(chapter)))
|
||||
}
|
||||
|
||||
/// Best-effort removal of staged page blobs after a rejected upload.
|
||||
async fn cleanup_staging(storage: &dyn Storage, staged: &[StagedImage]) {
|
||||
for page in staged {
|
||||
let _ = storage.delete(&page.staging_key).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Best-effort removal of already-finalized page blobs when the upload
|
||||
/// fails after some renames have landed.
|
||||
async fn cleanup_keys(storage: &dyn Storage, keys: &[String]) {
|
||||
for key in keys {
|
||||
let _ = storage.delete(key).await;
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, serde::Serialize)]
|
||||
struct PagesResponse {
|
||||
pages: Vec<Page>,
|
||||
|
||||
@@ -5,6 +5,13 @@
|
||||
//! The handler uses `Storage::get_stream` so a multi-MB page is piped to
|
||||
//! the client a chunk at a time instead of buffered server-side.
|
||||
//!
|
||||
//! **Thumbnails.** `?w=<px>` serves a width-bounded variant (grids ask for a
|
||||
//! small width so they download ~KB instead of the 1–5 MB original). The width
|
||||
//! snaps to a small allow-list so the number of cached derivatives stays
|
||||
//! bounded; the resized image is cached in storage under a `thumbs/w{W}/` prefix
|
||||
//! and regenerated on demand. Only JPEG/PNG sources are thumbnailed (encoders we
|
||||
//! ship); other formats fall back to the original.
|
||||
//!
|
||||
//! **Auth model — capability URLs by design.** This endpoint is
|
||||
//! deliberately unauthenticated: reads stay public per the project
|
||||
//! brief, and per-page authorisation would require either a per-request
|
||||
@@ -16,41 +23,250 @@
|
||||
//! would gate this endpoint behind a `Storage::owner_of(key)` check;
|
||||
//! the seam is intentional.
|
||||
|
||||
use std::io::Cursor;
|
||||
|
||||
use axum::body::Body;
|
||||
use axum::extract::{Path, State};
|
||||
use axum::extract::{Path, Query, State};
|
||||
use axum::http::{header, HeaderName};
|
||||
use axum::response::{IntoResponse, Response};
|
||||
use axum::routing::get;
|
||||
use axum::Router;
|
||||
use image::imageops::FilterType;
|
||||
use image::{ImageFormat, ImageReader};
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::app::AppState;
|
||||
use crate::error::AppResult;
|
||||
use crate::storage::StorageError;
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::storage::{Storage, StorageError};
|
||||
|
||||
/// Widths a thumbnail may be rendered at. A requested width snaps up to the
|
||||
/// smallest of these so the set of cached derivatives stays tiny.
|
||||
const ALLOWED_THUMB_WIDTHS: &[u32] = &[160, 320, 480, 640, 960];
|
||||
|
||||
/// Storage key prefix for cached thumbnails.
|
||||
const THUMB_PREFIX: &str = "thumbs";
|
||||
|
||||
/// Decode allocation cap (mirrors `analysis::ocr`): a tiny file declaring huge
|
||||
/// dimensions is rejected before the decoder allocates, not after (OOM guard).
|
||||
const MAX_THUMB_DECODE_PIXELS: u64 = 40_000_000;
|
||||
|
||||
pub fn routes() -> Router<AppState> {
|
||||
Router::new().route("/files/*key", get(serve))
|
||||
}
|
||||
|
||||
async fn serve(State(state): State<AppState>, Path(key): Path<String>) -> AppResult<Response> {
|
||||
let file = match state.storage.get_stream(&key).await {
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct ServeQuery {
|
||||
/// Requested thumbnail width in pixels; absent = serve the original.
|
||||
w: Option<String>,
|
||||
}
|
||||
|
||||
async fn serve(
|
||||
State(state): State<AppState>,
|
||||
Path(key): Path<String>,
|
||||
Query(q): Query<ServeQuery>,
|
||||
) -> AppResult<Response> {
|
||||
// Thumbnail request: only for source formats we can re-encode; anything
|
||||
// else falls through to serving the original.
|
||||
if let Some(width) = resolve_thumb_width(q.w.as_deref()) {
|
||||
if let Some(fmt) = thumb_format_for(&key) {
|
||||
return serve_thumbnail(&state, &key, width, fmt).await;
|
||||
}
|
||||
}
|
||||
serve_original(&state, &key).await
|
||||
}
|
||||
|
||||
async fn serve_original(state: &AppState, key: &str) -> AppResult<Response> {
|
||||
let file = match state.storage.get_stream(key).await {
|
||||
Ok(f) => f,
|
||||
Err(StorageError::NotFound) => return Err(crate::error::AppError::NotFound),
|
||||
Err(StorageError::NotFound) => return Err(AppError::NotFound),
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let ct = content_type_for(&key);
|
||||
// `nosniff` makes the contract explicit: the browser must trust the
|
||||
// Content-Type we declared (and that the magic-byte sniff at upload
|
||||
// time produced) instead of trying to detect HTML/JS in the body.
|
||||
// Belt-and-braces vs. polyglot files that survive the upload sniff.
|
||||
let headers = [
|
||||
(header::CONTENT_TYPE, ct.to_string()),
|
||||
(header::CONTENT_LENGTH, file.size_bytes.to_string()),
|
||||
Ok(image_response(
|
||||
state,
|
||||
content_type_for(key),
|
||||
file.size_bytes.to_string(),
|
||||
Body::from_stream(file.stream),
|
||||
))
|
||||
}
|
||||
|
||||
async fn serve_thumbnail(
|
||||
state: &AppState,
|
||||
key: &str,
|
||||
width: u32,
|
||||
fmt: ImageFormat,
|
||||
) -> AppResult<Response> {
|
||||
let derived = thumb_key(key, width);
|
||||
|
||||
// Serve the cached variant if it exists.
|
||||
match state.storage.get_stream(&derived).await {
|
||||
Ok(f) => {
|
||||
return Ok(image_response(
|
||||
state,
|
||||
content_type_for(key),
|
||||
f.size_bytes.to_string(),
|
||||
Body::from_stream(f.stream),
|
||||
));
|
||||
}
|
||||
Err(StorageError::NotFound) => {}
|
||||
Err(e) => return Err(e.into()),
|
||||
}
|
||||
|
||||
// Generate from the original.
|
||||
let original = match state.storage.get(key).await {
|
||||
Ok(b) => b,
|
||||
Err(StorageError::NotFound) => return Err(AppError::NotFound),
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
// Resizing is CPU-bound; keep it off the async worker threads.
|
||||
let thumb = match tokio::task::spawn_blocking(move || make_thumbnail(&original, width, fmt))
|
||||
.await
|
||||
.map_err(|e| AppError::Other(anyhow::anyhow!("thumbnail task join: {e}")))?
|
||||
{
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
// The blob's magic bytes passed upload's sniff but the body won't
|
||||
// decode (corrupt/truncated, or an encoder feature we don't support).
|
||||
// Don't 500 the reader — fall back to streaming the original, which
|
||||
// serves without decoding.
|
||||
tracing::warn!(key, error = %format!("{e:#}"), "thumbnail decode failed; serving original");
|
||||
return serve_original(state, key).await;
|
||||
}
|
||||
};
|
||||
|
||||
// Best-effort cache; a write failure just means we regenerate next time.
|
||||
let _ = state.storage.put(&derived, &thumb).await;
|
||||
|
||||
let len = thumb.len().to_string();
|
||||
Ok(image_response(
|
||||
state,
|
||||
content_type_for(key),
|
||||
len,
|
||||
Body::from(thumb),
|
||||
))
|
||||
}
|
||||
|
||||
/// Shared response builder for both the original and thumbnail paths.
|
||||
fn image_response(
|
||||
state: &AppState,
|
||||
content_type: &str,
|
||||
content_length: String,
|
||||
body: Body,
|
||||
) -> Response {
|
||||
let mut headers = vec![
|
||||
(header::CONTENT_TYPE, content_type.to_string()),
|
||||
(header::CONTENT_LENGTH, content_length),
|
||||
// `nosniff` makes the contract explicit: the browser must trust the
|
||||
// Content-Type we declared (and that the magic-byte sniff at upload
|
||||
// time produced) instead of trying to detect HTML/JS in the body.
|
||||
(
|
||||
HeaderName::from_static("x-content-type-options"),
|
||||
"nosniff".to_string(),
|
||||
),
|
||||
// Blobs are content-addressed by unguessable, immutable keys (a
|
||||
// re-upload mints new UUIDs), so a fetched page/cover never changes.
|
||||
// Cache it for a year and mark it `immutable` so browsers skip
|
||||
// revalidation entirely.
|
||||
//
|
||||
// BUT under PRIVATE_MODE these blobs are auth-gated, so they must NOT be
|
||||
// marked `public`: a shared cache / CDN would store the object and serve
|
||||
// it to unauthenticated clients. Use `private` so only the requesting
|
||||
// user's browser caches it.
|
||||
(
|
||||
header::CACHE_CONTROL,
|
||||
if state.auth.private_mode {
|
||||
"private, max-age=31536000, immutable".to_string()
|
||||
} else {
|
||||
"public, max-age=31536000, immutable".to_string()
|
||||
},
|
||||
),
|
||||
];
|
||||
Ok((headers, Body::from_stream(file.stream)).into_response())
|
||||
// Known image types render inline (covers/pages display in the reader). The
|
||||
// `application/octet-stream` fallback is a blob we couldn't type — it could
|
||||
// be crafted HTML/JS, so force a download rather than let the browser render
|
||||
// it inline. Belt to `nosniff`'s braces.
|
||||
if content_type == "application/octet-stream" {
|
||||
headers.push((
|
||||
header::CONTENT_DISPOSITION,
|
||||
"attachment".to_string(),
|
||||
));
|
||||
}
|
||||
(axum::response::AppendHeaders(headers), body).into_response()
|
||||
}
|
||||
|
||||
/// Parse and clamp a requested thumbnail width. Returns `None` for absent /
|
||||
/// unparseable / zero widths (serve the original); otherwise snaps the request
|
||||
/// up to the smallest allowed width (capped at the largest) so cached variants
|
||||
/// stay bounded.
|
||||
fn resolve_thumb_width(raw: Option<&str>) -> Option<u32> {
|
||||
let requested: u32 = raw?.trim().parse().ok()?;
|
||||
if requested == 0 {
|
||||
return None;
|
||||
}
|
||||
Some(
|
||||
ALLOWED_THUMB_WIDTHS
|
||||
.iter()
|
||||
.copied()
|
||||
.find(|&w| w >= requested)
|
||||
.unwrap_or_else(|| *ALLOWED_THUMB_WIDTHS.last().expect("non-empty")),
|
||||
)
|
||||
}
|
||||
|
||||
/// The re-encode format for a source key, or `None` when it isn't one we ship an
|
||||
/// encoder for (gif/avif → serve the original instead of a broken thumbnail).
|
||||
fn thumb_format_for(key: &str) -> Option<ImageFormat> {
|
||||
match content_type_for(key) {
|
||||
"image/jpeg" => Some(ImageFormat::Jpeg),
|
||||
"image/png" => Some(ImageFormat::Png),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// The storage key a width-`w` thumbnail of `key` is cached under.
|
||||
fn thumb_key(key: &str, width: u32) -> String {
|
||||
format!("{THUMB_PREFIX}/w{width}/{key}")
|
||||
}
|
||||
|
||||
/// Every cached thumbnail key for an original `key`, across all allowed widths.
|
||||
/// Used by the cover handlers to purge stale variants when a cover (whose key is
|
||||
/// reused, unlike content-addressed pages) is replaced or deleted.
|
||||
pub(crate) fn thumbnail_keys(key: &str) -> Vec<String> {
|
||||
ALLOWED_THUMB_WIDTHS
|
||||
.iter()
|
||||
.map(|&w| thumb_key(key, w))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Best-effort deletion of every cached thumbnail for `key`. Call after the
|
||||
/// underlying blob at `key` changes or is removed.
|
||||
pub(crate) async fn purge_thumbnails(storage: &dyn Storage, key: &str) {
|
||||
for derived in thumbnail_keys(key) {
|
||||
let _ = storage.delete(&derived).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode, downscale to `width` (aspect-preserving, never upscaling), and
|
||||
/// re-encode in `fmt`. Pure + synchronous so it runs under `spawn_blocking` and
|
||||
/// is unit-testable without a server.
|
||||
fn make_thumbnail(bytes: &[u8], width: u32, fmt: ImageFormat) -> anyhow::Result<Vec<u8>> {
|
||||
use anyhow::Context;
|
||||
let mut reader = ImageReader::new(Cursor::new(bytes))
|
||||
.with_guessed_format()
|
||||
.context("guess image format for thumbnail")?;
|
||||
let mut limits = image::Limits::default();
|
||||
limits.max_alloc = Some(MAX_THUMB_DECODE_PIXELS.saturating_mul(4));
|
||||
reader.limits(limits);
|
||||
let img = reader.decode().context("decode image for thumbnail")?;
|
||||
|
||||
// Only downscale; a source narrower than the target is served as-is.
|
||||
let out = if img.width() > width {
|
||||
img.resize(width, u32::MAX, FilterType::Lanczos3)
|
||||
} else {
|
||||
img
|
||||
};
|
||||
|
||||
let mut buf = Cursor::new(Vec::new());
|
||||
out.write_to(&mut buf, fmt).context("encode thumbnail")?;
|
||||
Ok(buf.into_inner())
|
||||
}
|
||||
|
||||
fn content_type_for(key: &str) -> &'static str {
|
||||
@@ -64,3 +280,68 @@ fn content_type_for(key: &str) -> &'static str {
|
||||
_ => "application/octet-stream",
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn resolve_thumb_width_snaps_and_validates() {
|
||||
// Absent / unparseable / zero → serve the original.
|
||||
assert_eq!(resolve_thumb_width(None), None);
|
||||
assert_eq!(resolve_thumb_width(Some("")), None);
|
||||
assert_eq!(resolve_thumb_width(Some("abc")), None);
|
||||
assert_eq!(resolve_thumb_width(Some("0")), None);
|
||||
// Snap up to the smallest allowed width.
|
||||
assert_eq!(resolve_thumb_width(Some("1")), Some(160));
|
||||
assert_eq!(resolve_thumb_width(Some("160")), Some(160));
|
||||
assert_eq!(resolve_thumb_width(Some("161")), Some(320));
|
||||
assert_eq!(resolve_thumb_width(Some("640")), Some(640));
|
||||
// Above the max → capped at the largest allowed width.
|
||||
assert_eq!(resolve_thumb_width(Some("5000")), Some(960));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thumb_format_only_for_encodable_sources() {
|
||||
assert_eq!(thumb_format_for("a/b/cover.jpg"), Some(ImageFormat::Jpeg));
|
||||
assert_eq!(thumb_format_for("a/b/cover.png"), Some(ImageFormat::Png));
|
||||
// Formats we can't re-encode fall back to the original.
|
||||
assert_eq!(thumb_format_for("a/b/cover.webp"), None);
|
||||
assert_eq!(thumb_format_for("a/b/cover.gif"), None);
|
||||
assert_eq!(thumb_format_for("a/b/cover.avif"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thumb_key_is_prefixed_by_width() {
|
||||
assert_eq!(
|
||||
thumb_key("mangas/x/cover.png", 320),
|
||||
"thumbs/w320/mangas/x/cover.png"
|
||||
);
|
||||
assert_eq!(thumbnail_keys("k.png").len(), ALLOWED_THUMB_WIDTHS.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn make_thumbnail_downscales_and_preserves_aspect() {
|
||||
// 100x50 red PNG → thumbnail width 40 → 40x20, still decodable PNG.
|
||||
let mut src = Cursor::new(Vec::new());
|
||||
image::RgbImage::from_pixel(100, 50, image::Rgb([255, 0, 0]))
|
||||
.write_to(&mut src, ImageFormat::Png)
|
||||
.unwrap();
|
||||
let out = make_thumbnail(src.get_ref(), 40, ImageFormat::Png).unwrap();
|
||||
let decoded = image::load_from_memory(&out).unwrap();
|
||||
assert_eq!(decoded.width(), 40);
|
||||
assert_eq!(decoded.height(), 20, "aspect ratio preserved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn make_thumbnail_does_not_upscale() {
|
||||
// A 30px-wide source requested at 40 stays 30 wide (no upscaling).
|
||||
let mut src = Cursor::new(Vec::new());
|
||||
image::RgbImage::from_pixel(30, 30, image::Rgb([0, 255, 0]))
|
||||
.write_to(&mut src, ImageFormat::Png)
|
||||
.unwrap();
|
||||
let out = make_thumbnail(src.get_ref(), 40, ImageFormat::Png).unwrap();
|
||||
let decoded = image::load_from_memory(&out).unwrap();
|
||||
assert_eq!(decoded.width(), 30);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ pub fn routes() -> Router<AppState> {
|
||||
.route("/mangas", get(list).post(create))
|
||||
.route("/mangas/:id", get(get_one).patch(update))
|
||||
.route("/mangas/:id/similar", get(list_similar))
|
||||
.route("/me/recommendations", get(list_recommendations))
|
||||
.route("/mangas/:id/cover", put(put_cover).delete(delete_cover))
|
||||
.route("/mangas/:id/tags", post(attach_tag))
|
||||
.route("/mangas/:id/tags/:tag_id", delete(detach_tag))
|
||||
@@ -156,7 +157,9 @@ async fn list(
|
||||
order,
|
||||
};
|
||||
let (items, total) = repo::manga::list_cards(&state.db, &q).await?;
|
||||
Ok(Json(PagedResponse::with_total(items, limit, offset, total)))
|
||||
Ok(Json(PagedResponse::with_optional_total(
|
||||
items, limit, offset, total,
|
||||
)))
|
||||
}
|
||||
|
||||
async fn get_one(
|
||||
@@ -188,6 +191,28 @@ async fn list_similar(
|
||||
Ok(Json(json!({ "items": items })))
|
||||
}
|
||||
|
||||
const RECOMMENDATIONS_LIMIT: i64 = 12;
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct RecommendationParams {
|
||||
#[serde(default)]
|
||||
pub limit: Option<i64>,
|
||||
}
|
||||
|
||||
/// `GET /api/v1/me/recommendations` — content-based "Recommended for you",
|
||||
/// ranked by tag overlap with the signed-in user's likes/bookmarks (minus
|
||||
/// dislikes). Returns `{ "items": [...] }` (a fixed top-N, like `/similar`);
|
||||
/// empty when the user has no taste signals yet.
|
||||
async fn list_recommendations(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
Query(params): Query<RecommendationParams>,
|
||||
) -> AppResult<Json<serde_json::Value>> {
|
||||
let limit = params.limit.unwrap_or(RECOMMENDATIONS_LIMIT).clamp(1, 50);
|
||||
let items = repo::manga::list_recommendations(&state.db, user.id, limit).await?;
|
||||
Ok(Json(json!({ "items": items })))
|
||||
}
|
||||
|
||||
/// `POST /api/v1/mangas` is multipart/form-data. Parts:
|
||||
///
|
||||
/// - `metadata` (required): JSON body matching `NewManga` — title, optional
|
||||
@@ -210,11 +235,17 @@ async fn create(
|
||||
while let Some(field) = next_field(&mut multipart).await? {
|
||||
match field.name() {
|
||||
Some("metadata") => {
|
||||
let bytes = read_field_bytes(field).await?;
|
||||
let bytes = crate::upload::read_capped(
|
||||
field,
|
||||
crate::upload::MAX_METADATA_BYTES,
|
||||
"metadata",
|
||||
)
|
||||
.await?;
|
||||
metadata = Some(parse_metadata_json(&bytes)?);
|
||||
}
|
||||
Some("cover") => {
|
||||
let bytes = read_field_bytes(field).await?.to_vec();
|
||||
let bytes =
|
||||
crate::upload::read_capped(field, state.upload.max_file_bytes, "cover").await?;
|
||||
cover = Some(parse_image(bytes, state.upload.max_file_bytes, "cover")?);
|
||||
}
|
||||
_ => continue,
|
||||
@@ -255,8 +286,8 @@ async fn create(
|
||||
)
|
||||
.await?;
|
||||
|
||||
let author_refs = repo::author::set_for_manga(&mut *tx, manga.id, &authors).await?;
|
||||
repo::genre::set_for_manga(&mut *tx, manga.id, &metadata.genre_ids).await?;
|
||||
let author_refs = repo::author::set_for_manga(&mut tx, manga.id, &authors).await?;
|
||||
repo::genre::set_for_manga(&mut tx, manga.id, &metadata.genre_ids).await?;
|
||||
|
||||
if let Some(img) = cover {
|
||||
let key = format!("mangas/{}/cover.{}", manga.id, img.ext);
|
||||
@@ -321,7 +352,7 @@ async fn update(
|
||||
|
||||
let mut tx = state.db.begin().await?;
|
||||
let _updated = repo::manga::update_basics(
|
||||
&mut *tx,
|
||||
&mut tx,
|
||||
id,
|
||||
patch.title.as_deref().map(str::trim),
|
||||
patch.status.as_deref().map(str::trim),
|
||||
@@ -331,10 +362,10 @@ async fn update(
|
||||
)
|
||||
.await?;
|
||||
if let Some(ref names) = authors_owned {
|
||||
repo::author::set_for_manga(&mut *tx, id, names).await?;
|
||||
repo::author::set_for_manga(&mut tx, id, names).await?;
|
||||
}
|
||||
if let Some(ref ids) = patch.genre_ids {
|
||||
repo::genre::set_for_manga(&mut *tx, id, ids).await?;
|
||||
repo::genre::set_for_manga(&mut tx, id, ids).await?;
|
||||
}
|
||||
tx.commit().await?;
|
||||
|
||||
@@ -362,7 +393,8 @@ async fn put_cover(
|
||||
let mut cover: Option<UploadedImage> = None;
|
||||
while let Some(field) = next_field(&mut multipart).await? {
|
||||
if field.name() == Some("cover") {
|
||||
let bytes = read_field_bytes(field).await?.to_vec();
|
||||
let bytes =
|
||||
crate::upload::read_capped(field, state.upload.max_file_bytes, "cover").await?;
|
||||
cover = Some(parse_image(bytes, state.upload.max_file_bytes, "cover")?);
|
||||
}
|
||||
}
|
||||
@@ -377,6 +409,17 @@ async fn put_cover(
|
||||
let old_key = repo::manga::get(&state.db, id).await?.cover_image_path;
|
||||
let new_key = format!("mangas/{}/cover.{}", id, img.ext);
|
||||
state.storage.put(&new_key, &img.bytes).await?;
|
||||
// Cover keys are reused (mangas/{id}/cover.{ext}), so a same-extension
|
||||
// replacement overwrites the blob at an existing key — drop any thumbnails
|
||||
// cached for it so we don't serve a stale variant of the old cover.
|
||||
crate::api::files::purge_thumbnails(state.storage.as_ref(), &new_key).await;
|
||||
|
||||
// Commit the DB pointer to the new blob BEFORE removing the old one. If we
|
||||
// deleted the old blob first and this write then failed, the row would point
|
||||
// at a deleted blob (a cover that 404s) and the new blob would be orphaned.
|
||||
// Ordering it first means a failure here leaves the row pointing at the
|
||||
// still-present old blob; only a harmless orphan (new blob) can result.
|
||||
repo::manga::set_cover_image_path(&state.db, id, &new_key, img.bytes.len() as i64).await?;
|
||||
|
||||
if let Some(prev) = old_key.as_deref() {
|
||||
if prev != new_key {
|
||||
@@ -387,10 +430,10 @@ async fn put_cover(
|
||||
Ok(()) | Err(StorageError::NotFound) => {}
|
||||
Err(e) => return Err(e.into()),
|
||||
}
|
||||
// Old key's cached thumbnails are now orphaned.
|
||||
crate::api::files::purge_thumbnails(state.storage.as_ref(), prev).await;
|
||||
}
|
||||
}
|
||||
|
||||
repo::manga::set_cover_image_path(&state.db, id, &new_key, img.bytes.len() as i64).await?;
|
||||
Ok(Json(repo::manga::get_detail(&state.db, id).await?))
|
||||
}
|
||||
|
||||
@@ -408,11 +451,15 @@ async fn delete_cover(
|
||||
}
|
||||
require_can_edit(&state, id, user.id, admin_via_session(&session)).await?;
|
||||
if let Some(key) = repo::manga::get(&state.db, id).await?.cover_image_path {
|
||||
// Clear the DB pointer BEFORE deleting the blob, so a storage-delete
|
||||
// failure can't leave the row pointing at a removed blob. A failure
|
||||
// after the clear only orphans the blob (harmless).
|
||||
repo::manga::clear_cover_image_path(&state.db, id).await?;
|
||||
match state.storage.delete(&key).await {
|
||||
Ok(()) | Err(StorageError::NotFound) => {}
|
||||
Err(e) => return Err(e.into()),
|
||||
}
|
||||
repo::manga::clear_cover_image_path(&state.db, id).await?;
|
||||
crate::api::files::purge_thumbnails(state.storage.as_ref(), &key).await;
|
||||
}
|
||||
Ok(Json(repo::manga::get_detail(&state.db, id).await?))
|
||||
}
|
||||
@@ -643,13 +690,7 @@ pub(crate) async fn next_field(
|
||||
.map_err(map_multipart_error)
|
||||
}
|
||||
|
||||
pub(crate) async fn read_field_bytes(
|
||||
field: axum::extract::multipart::Field<'_>,
|
||||
) -> AppResult<axum::body::Bytes> {
|
||||
field.bytes().await.map_err(map_multipart_error)
|
||||
}
|
||||
|
||||
fn map_multipart_error(e: axum::extract::multipart::MultipartError) -> AppError {
|
||||
pub(crate) fn map_multipart_error(e: axum::extract::multipart::MultipartError) -> AppError {
|
||||
let status = e.status();
|
||||
if status == StatusCode::PAYLOAD_TOO_LARGE {
|
||||
AppError::PayloadTooLarge("upload exceeds the request size limit".into())
|
||||
|
||||
@@ -11,6 +11,7 @@ pub mod history;
|
||||
pub mod mangas;
|
||||
pub mod page_tags;
|
||||
pub mod pagination;
|
||||
pub mod reactions;
|
||||
pub mod tags;
|
||||
|
||||
use axum::Router;
|
||||
@@ -31,5 +32,6 @@ pub fn routes() -> Router<AppState> {
|
||||
.merge(collections::routes())
|
||||
.merge(page_tags::routes())
|
||||
.merge(history::routes())
|
||||
.merge(reactions::routes())
|
||||
.merge(admin::routes())
|
||||
}
|
||||
|
||||
@@ -392,10 +392,9 @@ pub struct AggregateParams {
|
||||
pub limit: i64,
|
||||
#[serde(default)]
|
||||
pub offset: i64,
|
||||
/// Reserved for the planned OCR text-search input. Accepted on
|
||||
/// the wire so adding OCR later won't break the API shape, but
|
||||
/// rejected with 501 `text_search_not_yet_supported` if non-empty
|
||||
/// until the backend supports it.
|
||||
/// OCR text filter. When non-empty, only pages whose analysis
|
||||
/// `search_doc` matches the query (`plainto_tsquery`) are aggregated.
|
||||
/// Blank/absent ⇒ tag-only aggregation.
|
||||
#[serde(default)]
|
||||
pub text: Option<String>,
|
||||
}
|
||||
@@ -411,32 +410,17 @@ fn parse_order(raw: Option<&str>) -> AppResult<Order> {
|
||||
}
|
||||
}
|
||||
|
||||
fn ensure_text_unsupported(text: Option<&str>) -> AppResult<()> {
|
||||
// Future OCR search will plug in here. Until then, return a
|
||||
// distinct code (`text_search_not_yet_supported`) so clients can
|
||||
// detect "feature pending" vs. a generic 4xx — the code is the
|
||||
// wire contract, not the message.
|
||||
if text.is_some_and(|s| !s.trim().is_empty()) {
|
||||
return Err(AppError::NotImplemented {
|
||||
code: "text_search_not_yet_supported",
|
||||
message: "text search is reserved for the planned OCR input but not yet supported",
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn list_chapters_for_tag(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
Query(params): Query<AggregateParams>,
|
||||
) -> AppResult<Json<PagedResponse<TaggedChapterAggregate>>> {
|
||||
ensure_text_unsupported(params.text.as_deref())?;
|
||||
let tag = normalize_tag(¶ms.tag)?;
|
||||
let order = parse_order(params.order.as_deref())?;
|
||||
let limit = params.limit.clamp(1, 200);
|
||||
let offset = params.offset.max(0);
|
||||
let (items, total) = repo::page_tag::aggregate_chapters_for_tag(
|
||||
&state.db, user.id, &tag, order, limit, offset,
|
||||
&state.db, user.id, &tag, order, limit, offset, params.text.as_deref(),
|
||||
)
|
||||
.await?;
|
||||
Ok(Json(PagedResponse::with_total(items, limit, offset, total)))
|
||||
@@ -447,13 +431,12 @@ async fn list_mangas_for_tag(
|
||||
CurrentUser(user): CurrentUser,
|
||||
Query(params): Query<AggregateParams>,
|
||||
) -> AppResult<Json<PagedResponse<TaggedMangaAggregate>>> {
|
||||
ensure_text_unsupported(params.text.as_deref())?;
|
||||
let tag = normalize_tag(¶ms.tag)?;
|
||||
let order = parse_order(params.order.as_deref())?;
|
||||
let limit = params.limit.clamp(1, 200);
|
||||
let offset = params.offset.max(0);
|
||||
let (items, total) = repo::page_tag::aggregate_mangas_for_tag(
|
||||
&state.db, user.id, &tag, order, limit, offset,
|
||||
&state.db, user.id, &tag, order, limit, offset, params.text.as_deref(),
|
||||
)
|
||||
.await?;
|
||||
Ok(Json(PagedResponse::with_total(items, limit, offset, total)))
|
||||
|
||||
@@ -34,4 +34,18 @@ impl<T> PagedResponse<T> {
|
||||
page: PageInfo { limit, offset, total: Some(total) },
|
||||
}
|
||||
}
|
||||
|
||||
/// For handlers that compute `total` only on some pages (e.g. the first)
|
||||
/// and leave it `None` elsewhere.
|
||||
pub fn with_optional_total(
|
||||
items: Vec<T>,
|
||||
limit: i64,
|
||||
offset: i64,
|
||||
total: Option<i64>,
|
||||
) -> Self {
|
||||
Self {
|
||||
items,
|
||||
page: PageInfo { limit, offset, total },
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
69
backend/src/api/reactions.rs
Normal file
69
backend/src/api/reactions.rs
Normal file
@@ -0,0 +1,69 @@
|
||||
//! Manga reactions (like/dislike) — a private, per-user taste signal.
|
||||
//! Writes require auth; the read is scoped under `/me/` so the URL can't be
|
||||
//! used to peek at another user's reactions.
|
||||
|
||||
use axum::extract::{Path, State};
|
||||
use axum::http::StatusCode;
|
||||
use axum::routing::{get, put};
|
||||
use axum::{Json, Router};
|
||||
use serde::Deserialize;
|
||||
use serde_json::json;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::app::AppState;
|
||||
use crate::auth::extractor::CurrentUser;
|
||||
use crate::domain::reaction::{MangaReaction, Reaction};
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::repo;
|
||||
|
||||
pub fn routes() -> Router<AppState> {
|
||||
Router::new()
|
||||
.route(
|
||||
"/mangas/:id/reaction",
|
||||
put(set_reaction).delete(clear_reaction),
|
||||
)
|
||||
.route("/me/reactions/:manga_id", get(get_reaction))
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct SetReactionBody {
|
||||
pub reaction: String,
|
||||
}
|
||||
|
||||
async fn set_reaction(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
Path(manga_id): Path<Uuid>,
|
||||
Json(body): Json<SetReactionBody>,
|
||||
) -> AppResult<Json<MangaReaction>> {
|
||||
// Validate against the closed vocabulary here so a bad value is a clean
|
||||
// 422 rather than relying on the DB CHECK to surface as a 500.
|
||||
let reaction = Reaction::parse(&body.reaction).ok_or_else(|| AppError::ValidationFailed {
|
||||
message: "reaction must be 'like' or 'dislike'".into(),
|
||||
details: json!({ "reaction": "must be 'like' or 'dislike'" }),
|
||||
})?;
|
||||
// Unknown manga → 404 via the FK-violation mapping in repo::reaction.
|
||||
repo::reaction::upsert(&state.db, user.id, manga_id, reaction).await?;
|
||||
Ok(Json(MangaReaction {
|
||||
manga_id,
|
||||
reaction: Some(reaction),
|
||||
}))
|
||||
}
|
||||
|
||||
async fn clear_reaction(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
Path(manga_id): Path<Uuid>,
|
||||
) -> AppResult<StatusCode> {
|
||||
repo::reaction::clear(&state.db, user.id, manga_id).await?;
|
||||
Ok(StatusCode::NO_CONTENT)
|
||||
}
|
||||
|
||||
async fn get_reaction(
|
||||
State(state): State<AppState>,
|
||||
CurrentUser(user): CurrentUser,
|
||||
Path(manga_id): Path<Uuid>,
|
||||
) -> AppResult<Json<MangaReaction>> {
|
||||
let reaction = repo::reaction::get(&state.db, user.id, manga_id).await?;
|
||||
Ok(Json(MangaReaction { manga_id, reaction }))
|
||||
}
|
||||
@@ -277,9 +277,15 @@ impl DaemonReloader for Supervisors {
|
||||
}
|
||||
}
|
||||
|
||||
/// How often the background reaper sweeps expired sessions. Hourly is ample:
|
||||
/// the sweep is a single indexed DELETE and expired rows are already invisible
|
||||
/// to auth, so this is purely storage hygiene.
|
||||
const SESSION_GC_INTERVAL: std::time::Duration = std::time::Duration::from_secs(3600);
|
||||
|
||||
pub async fn build(config: Config) -> anyhow::Result<AppHandle> {
|
||||
let db = PgPoolOptions::new()
|
||||
.max_connections(10)
|
||||
.max_connections(config.db.max_connections)
|
||||
.acquire_timeout(config.db.acquire_timeout)
|
||||
.connect(&config.database_url)
|
||||
.await?;
|
||||
sqlx::migrate!("./migrations").run(&db).await?;
|
||||
@@ -336,6 +342,27 @@ pub async fn build(config: Config) -> anyhow::Result<AppHandle> {
|
||||
tracing::info!("analysis worker disabled");
|
||||
}
|
||||
|
||||
// Periodic reaper for lapsed sessions. `find_active` already ignores
|
||||
// expired rows, so this only reclaims storage — without it the table grows
|
||||
// unbounded as sessions lapse. Detached and best-effort: a failed sweep is
|
||||
// logged and retried next tick. Runs regardless of crawler/analysis config
|
||||
// since sessions exist in every deployment.
|
||||
{
|
||||
let db = db.clone();
|
||||
tokio::spawn(async move {
|
||||
let mut ticker = tokio::time::interval(SESSION_GC_INTERVAL);
|
||||
ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||
loop {
|
||||
ticker.tick().await;
|
||||
match repo::session::delete_expired(&db).await {
|
||||
Ok(0) => {}
|
||||
Ok(n) => tracing::info!(reaped = n, "session gc: removed expired sessions"),
|
||||
Err(e) => tracing::warn!(?e, "session gc sweep failed; retrying next tick"),
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
let auth_limiter = Arc::new(AuthRateLimiter::new(config.auth.rate_limit));
|
||||
let state = AppState {
|
||||
db,
|
||||
@@ -429,46 +456,98 @@ async fn spawn_analysis_daemon(
|
||||
),
|
||||
Err(e) => tracing::warn!(?e, "analysis: reclaim_orphaned at startup failed"),
|
||||
}
|
||||
let http = reqwest::Client::builder()
|
||||
.timeout(cfg.request_timeout)
|
||||
// Refuse to honour ambient HTTP_PROXY / HTTPS_PROXY container env.
|
||||
// The vision call carries an env-managed bearer token + page image
|
||||
// bytes; a stray upstream proxy would exfiltrate both. Mirrors the
|
||||
// crawler client's `.no_proxy()` (see `spawn_crawler_daemon`).
|
||||
.no_proxy()
|
||||
.build()
|
||||
.context("build analysis http client")?;
|
||||
let vision = crate::analysis::vision::VisionClient::new(http, cfg);
|
||||
let dispatcher = Arc::new(crate::analysis::daemon::RealAnalyzeDispatcher {
|
||||
db: db.clone(),
|
||||
storage,
|
||||
vision,
|
||||
model: cfg.model.clone(),
|
||||
max_image_bytes: cfg.max_image_bytes,
|
||||
});
|
||||
// When a readiness URL is configured, gate leasing on it so an
|
||||
// autoscaler that idle-stops the vision container never lets a job burn
|
||||
// its retries. A dedicated short-timeout client keeps the probe snappy
|
||||
// and independent of the (long) per-request analysis timeout.
|
||||
let readiness: Option<Arc<dyn crate::analysis::daemon::VisionReadiness>> =
|
||||
match &cfg.vision_health_url {
|
||||
Some(url) if !url.is_empty() => {
|
||||
let probe = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(5))
|
||||
// Same reasoning as the main analysis client: do not
|
||||
// honour ambient HTTP_PROXY env. The readiness probe is
|
||||
// unauthenticated but a hostile upstream still gets a
|
||||
// useful side-channel on backend uptime + vision health.
|
||||
.no_proxy()
|
||||
.build()
|
||||
.context("build vision readiness http client")?;
|
||||
Some(Arc::new(crate::analysis::daemon::HttpVisionReadiness {
|
||||
http: probe,
|
||||
health_url: url.clone(),
|
||||
}))
|
||||
}
|
||||
_ => None,
|
||||
};
|
||||
// Pick the engine. The OCR backend runs in-process (no network, no
|
||||
// readiness gate); the vision backend talks to a local LLM server.
|
||||
let (dispatcher, readiness): (
|
||||
Arc<dyn crate::analysis::daemon::AnalyzeDispatcher>,
|
||||
Option<Arc<dyn crate::analysis::daemon::VisionReadiness>>,
|
||||
) = match cfg.effective_backend() {
|
||||
crate::config::AnalysisBackend::Ocr => {
|
||||
// Load the `.rten` models once; a bad path is a loud boot error.
|
||||
let engine = crate::analysis::ocr::OcrsEngine::from_model_paths(
|
||||
&cfg.ocr_detection_model,
|
||||
&cfg.ocr_recognition_model,
|
||||
cfg.ocr_max_decode_pixels,
|
||||
)
|
||||
.context("build ocrs engine")?;
|
||||
// Cap concurrent CPU-bound OCR runs across all workers so a high
|
||||
// ANALYSIS_WORKERS can't oversubscribe the blocking pool.
|
||||
let cores = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1);
|
||||
let permits =
|
||||
crate::analysis::ocr::ocr_concurrency_limit(cfg.workers, cores);
|
||||
let dispatcher = Arc::new(crate::analysis::ocr::OcrAnalyzeDispatcher {
|
||||
db: db.clone(),
|
||||
storage,
|
||||
engine: Arc::new(engine),
|
||||
max_image_bytes: cfg.max_image_bytes,
|
||||
ocr_permits: Arc::new(tokio::sync::Semaphore::new(permits)),
|
||||
});
|
||||
// In-process engine is always ready — no gate.
|
||||
(dispatcher, None)
|
||||
}
|
||||
crate::config::AnalysisBackend::Vision => {
|
||||
let http = reqwest::Client::builder()
|
||||
.timeout(cfg.request_timeout)
|
||||
// Refuse to honour ambient HTTP_PROXY / HTTPS_PROXY container
|
||||
// env. The vision call carries an env-managed bearer token +
|
||||
// page image bytes; a stray upstream proxy would exfiltrate
|
||||
// both. Mirrors the crawler client's `.no_proxy()` (see
|
||||
// `spawn_crawler_daemon`).
|
||||
.no_proxy()
|
||||
// Re-validate redirect hops so a hostile/compromised vision
|
||||
// endpoint can't 302 the bearer token + page bytes into the
|
||||
// deployment's internal network. No allowlist here (the
|
||||
// endpoint is a single admin-configured URL), so the policy
|
||||
// enforces scheme + private-IP only.
|
||||
.redirect(crate::crawler::safety::public_redirect_policy())
|
||||
// NOTE: deliberately no `.dns_resolver(safe_dns_resolver())`.
|
||||
// The vision endpoint is a single operator-configured internal
|
||||
// service (e.g. `mangalord-vision`) that legitimately resolves
|
||||
// to a private Docker IP; a private-IP-rejecting resolver drops
|
||||
// every call. Redirect hops stay guarded by the policy above,
|
||||
// and the URL is operator-set, not attacker-controlled input.
|
||||
.build()
|
||||
.context("build analysis http client")?;
|
||||
let vision = crate::analysis::vision::VisionClient::new(http, cfg);
|
||||
let dispatcher = Arc::new(crate::analysis::daemon::RealAnalyzeDispatcher {
|
||||
db: db.clone(),
|
||||
storage,
|
||||
vision,
|
||||
model: cfg.model.clone(),
|
||||
max_image_bytes: cfg.max_image_bytes,
|
||||
});
|
||||
// When a readiness URL is configured, gate leasing on it so an
|
||||
// autoscaler that idle-stops the vision container never lets a job
|
||||
// burn its retries. A dedicated short-timeout client keeps the
|
||||
// probe snappy and independent of the (long) per-request timeout.
|
||||
let readiness: Option<Arc<dyn crate::analysis::daemon::VisionReadiness>> =
|
||||
match &cfg.vision_health_url {
|
||||
Some(url) if !url.is_empty() => {
|
||||
let probe = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(5))
|
||||
// Same reasoning as the main analysis client: do
|
||||
// not honour ambient HTTP_PROXY env. The readiness
|
||||
// probe is unauthenticated but a hostile upstream
|
||||
// still gets a useful side-channel on backend
|
||||
// uptime + vision health.
|
||||
.no_proxy()
|
||||
.redirect(crate::crawler::safety::public_redirect_policy())
|
||||
// No private-IP resolver: same operator-configured
|
||||
// internal vision host as the analysis client above.
|
||||
.build()
|
||||
.context("build vision readiness http client")?;
|
||||
Some(Arc::new(crate::analysis::daemon::HttpVisionReadiness {
|
||||
http: probe,
|
||||
health_url: url.clone(),
|
||||
}))
|
||||
}
|
||||
_ => None,
|
||||
};
|
||||
(dispatcher, readiness)
|
||||
}
|
||||
};
|
||||
let handle = crate::analysis::daemon::spawn(
|
||||
db,
|
||||
CancellationToken::new(),
|
||||
@@ -480,7 +559,21 @@ async fn spawn_analysis_daemon(
|
||||
readiness,
|
||||
},
|
||||
);
|
||||
tracing::info!(workers = cfg.workers, model = %cfg.model, "analysis worker daemon started");
|
||||
// Log the *effective* backend (what the worker actually dispatches
|
||||
// through), not the raw `cfg.backend` — with vision dormant they diverge
|
||||
// when an operator requests `vision`, and a misleading line here was a
|
||||
// real observability footgun. The model label tracks the effective engine.
|
||||
let effective_backend = cfg.effective_backend();
|
||||
let effective_model: &str = match effective_backend {
|
||||
crate::config::AnalysisBackend::Ocr => crate::analysis::ocr::OCR_MODEL_LABEL,
|
||||
crate::config::AnalysisBackend::Vision => &cfg.model,
|
||||
};
|
||||
tracing::info!(
|
||||
workers = cfg.workers,
|
||||
backend = ?effective_backend,
|
||||
model = %effective_model,
|
||||
"analysis worker daemon started"
|
||||
);
|
||||
Ok(handle)
|
||||
}
|
||||
|
||||
@@ -500,6 +593,10 @@ async fn spawn_crawler_daemon(
|
||||
cfg: &CrawlerConfig,
|
||||
analysis_enabled: Arc<AtomicBool>,
|
||||
) -> anyhow::Result<SpawnedDaemon> {
|
||||
// Publish the opt-in browser SSRF-interception toggle so every headless
|
||||
// navigation (via `intercept::open_page`) honours it. Off by default.
|
||||
crate::crawler::intercept::set_enabled(cfg.ssrf_intercept);
|
||||
|
||||
// Reqwest client with a shared cookie jar so CDN image fetches include
|
||||
// PHPSESSID. The same `Arc<Jar>` is held by the SessionController, so a
|
||||
// runtime session refresh rewrites it in place. Initial value: a
|
||||
@@ -520,6 +617,13 @@ async fn spawn_crawler_daemon(
|
||||
let mut http_builder = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(30))
|
||||
.no_proxy()
|
||||
// Re-validate every redirect hop against the download allowlist:
|
||||
// reqwest's default policy follows up to 10 redirects, and
|
||||
// `is_safe_url` only guards the initial URL, so an allowlisted CDN
|
||||
// 302ing to a private IP would otherwise be followed (SSRF).
|
||||
.redirect(crate::crawler::safety::safe_redirect_policy(
|
||||
cfg.download_allowlist.clone(),
|
||||
))
|
||||
.cookie_provider(Arc::clone(&cookie_jar));
|
||||
if let Some(ua) = &cfg.user_agent {
|
||||
http_builder = http_builder.user_agent(ua);
|
||||
@@ -528,6 +632,17 @@ async fn spawn_crawler_daemon(
|
||||
http_builder = http_builder
|
||||
.proxy(reqwest::Proxy::all(proxy).with_context(|| format!("parse proxy: {proxy}"))?);
|
||||
}
|
||||
// DNS-rebinding guard: reject hosts that resolve to a private/internal IP,
|
||||
// complementing the string-level allowlist check which can't see
|
||||
// post-resolution addresses. Attached on the direct path AND on http(s)
|
||||
// proxies (reqwest resolves the target itself there). Skipped only for SOCKS
|
||||
// proxies, where the proxy — not reqwest — resolves the target, so the only
|
||||
// name this resolver would see is the proxy's OWN host (legitimately on a
|
||||
// private Docker IP, e.g. `tor` → 172.x); attaching it there rejected every
|
||||
// fetch for zero gain. See `should_attach_safe_resolver`.
|
||||
if crate::crawler::safety::should_attach_safe_resolver(cfg.proxy.as_deref()) {
|
||||
http_builder = http_builder.dns_resolver(crate::crawler::safety::safe_dns_resolver());
|
||||
}
|
||||
let http = http_builder.build().context("build crawler reqwest")?;
|
||||
|
||||
let mut rate = HostRateLimiters::new(std::time::Duration::from_millis(cfg.rate_ms));
|
||||
@@ -651,6 +766,7 @@ async fn spawn_crawler_daemon(
|
||||
rate: Arc::clone(&rate),
|
||||
download_allowlist: cfg.download_allowlist.clone(),
|
||||
max_image_bytes: cfg.max_image_bytes,
|
||||
max_images_per_chapter: cfg.max_images_per_chapter,
|
||||
analysis_enabled,
|
||||
transient_failures: Arc::new(AtomicU32::new(0)),
|
||||
restart_threshold: cfg.browser_restart_threshold,
|
||||
@@ -667,6 +783,7 @@ async fn spawn_crawler_daemon(
|
||||
rate: Arc::clone(&rate),
|
||||
download_allowlist: cfg.download_allowlist.clone(),
|
||||
max_image_bytes: cfg.max_image_bytes,
|
||||
max_images_per_chapter: cfg.max_images_per_chapter,
|
||||
tor: tor.as_ref().map(Arc::clone),
|
||||
});
|
||||
|
||||
@@ -853,6 +970,8 @@ struct RealChapterDispatcher {
|
||||
rate: Arc<HostRateLimiters>,
|
||||
download_allowlist: DownloadAllowlist,
|
||||
max_image_bytes: usize,
|
||||
/// Per-chapter image-count cap (see `CrawlerConfig::max_images_per_chapter`).
|
||||
max_images_per_chapter: usize,
|
||||
/// Enqueue `analyze_page` jobs for freshly-crawled pages. Shared gate
|
||||
/// (read live) so toggling analysis at runtime takes effect without a
|
||||
/// crawler respawn. Mirrors the analysis enable setting.
|
||||
@@ -897,7 +1016,17 @@ impl ChapterDispatcher for RealChapterDispatcher {
|
||||
pages_done: 0,
|
||||
pages_total: None,
|
||||
});
|
||||
let lease = self.browser_manager.acquire().await?;
|
||||
let lease = match self.browser_manager.acquire().await {
|
||||
Ok(l) => l,
|
||||
Err(e) => {
|
||||
// Browser down / mid-restart: defer the job WITHOUT
|
||||
// burning an attempt (the daemon releases it back to
|
||||
// pending) rather than counting an infrastructure
|
||||
// outage as a per-job failure.
|
||||
tracing::warn!(error = ?e, "dispatch: browser unavailable — deferring job");
|
||||
return Ok(SyncOutcome::BrowserUnavailable);
|
||||
}
|
||||
};
|
||||
let result = content::sync_chapter_content(
|
||||
&lease,
|
||||
&self.db,
|
||||
@@ -910,6 +1039,7 @@ impl ChapterDispatcher for RealChapterDispatcher {
|
||||
false,
|
||||
&self.download_allowlist,
|
||||
self.max_image_bytes,
|
||||
self.max_images_per_chapter,
|
||||
self.tor.as_deref(),
|
||||
Some(&self.status),
|
||||
self.analysis_enabled.load(Ordering::Relaxed),
|
||||
@@ -967,7 +1097,15 @@ impl ChapterDispatcher for RealChapterDispatcher {
|
||||
// Scope the lease so it (and the borrowing FetchContext) drop
|
||||
// before any browser-restart handling in the match below.
|
||||
let result = {
|
||||
let lease = self.browser_manager.acquire().await?;
|
||||
let lease = match self.browser_manager.acquire().await {
|
||||
Ok(l) => l,
|
||||
Err(e) => {
|
||||
// See the SyncChapterContent arm: defer without
|
||||
// burning an attempt when the browser is unavailable.
|
||||
tracing::warn!(error = ?e, "dispatch: browser unavailable — deferring job");
|
||||
return Ok(SyncOutcome::BrowserUnavailable);
|
||||
}
|
||||
};
|
||||
let ctx = crate::crawler::source::FetchContext {
|
||||
browser: &lease,
|
||||
rate: &self.rate,
|
||||
@@ -1201,8 +1339,8 @@ fn parse_origin(raw: &str) -> Option<String> {
|
||||
let host = url.host_str()?;
|
||||
let scheme = url.scheme();
|
||||
let port_str = match (url.port(), scheme) {
|
||||
(Some(p), "http") if p == 80 => String::new(),
|
||||
(Some(p), "https") if p == 443 => String::new(),
|
||||
(Some(80), "http") => String::new(),
|
||||
(Some(443), "https") => String::new(),
|
||||
(Some(p), _) => format!(":{p}"),
|
||||
(None, _) => String::new(),
|
||||
};
|
||||
@@ -1440,25 +1578,32 @@ mod tests {
|
||||
// Bind to a local so the TempDir lives for the rest of the test.
|
||||
// `tempfile::tempdir().unwrap().path()` would drop the TempDir
|
||||
// at end-of-expression and `LocalStorage` would hold a path to
|
||||
// a deleted directory. (Today this is fine because the dispatch
|
||||
// fails before storage is touched, but it makes the test fragile
|
||||
// to any future code rearrangement.)
|
||||
// a deleted directory.
|
||||
let storage_dir = tempfile::tempdir().unwrap();
|
||||
let storage: Arc<dyn Storage> =
|
||||
Arc::new(LocalStorage::new(storage_dir.path()));
|
||||
let mut cfg = crate::config::AnalysisConfig::default();
|
||||
cfg.workers = 1;
|
||||
cfg.job_timeout = Duration::from_secs(1);
|
||||
// The worker always runs OCR now (vision is dormant — see
|
||||
// `effective_backend`), and the `.rten` models aren't shipped to unit
|
||||
// CI. Point the engine at a path that can't exist so the *engine build*
|
||||
// fails deterministically. Reclaim runs at the very top of
|
||||
// `spawn_analysis_daemon`, before — and independently of — engine
|
||||
// readiness, so the row must still be reclaimed even though spawn
|
||||
// returns Err. That's exactly the regression this test guards (an
|
||||
// analysis-only deploy must reclaim orphaned leases at startup).
|
||||
let cfg = crate::config::AnalysisConfig {
|
||||
ocr_detection_model: "/nonexistent/text-detection.rten".to_string(),
|
||||
ocr_recognition_model: "/nonexistent/text-recognition.rten".to_string(),
|
||||
workers: 1,
|
||||
job_timeout: Duration::from_secs(1),
|
||||
..Default::default()
|
||||
};
|
||||
let events = Arc::new(crate::analysis::events::AnalysisEvents::new());
|
||||
|
||||
let handle = spawn_analysis_daemon(pool.clone(), storage, &cfg, events)
|
||||
.await
|
||||
.expect("spawn");
|
||||
// Immediately shut down — we're only here to prove reclaim ran.
|
||||
// The workers may briefly pick up the now-pending row; that's
|
||||
// fine, but we cancel before any real dispatch (the LocalStorage
|
||||
// key doesn't exist, so a dispatch would fail anyway).
|
||||
handle.shutdown().await;
|
||||
let spawned = spawn_analysis_daemon(pool.clone(), storage, &cfg, events).await;
|
||||
assert!(
|
||||
spawned.is_err(),
|
||||
"engine build must fail with a missing model path"
|
||||
);
|
||||
|
||||
// The reclaim must have moved the row back to pending with the
|
||||
// attempt refunded (attempts goes from 1 → 0). Reaching it on the
|
||||
|
||||
@@ -70,6 +70,45 @@ impl FromRequestParts<AppState> for CurrentUser {
|
||||
}
|
||||
}
|
||||
|
||||
/// Client IP for per-IP auth rate limiting. Resolves to the first hop of
|
||||
/// `X-Forwarded-For` **only** when [`crate::config::AuthConfig::trusted_proxy`]
|
||||
/// is set (the backend is behind a proxy that overrides the header — the
|
||||
/// compose deploy). Otherwise `None`, so the limiter uses its shared bucket.
|
||||
/// Never fails: a missing or malformed header simply yields `None`.
|
||||
pub struct ClientIp(pub Option<std::net::IpAddr>);
|
||||
|
||||
/// Parse the client IP from an `X-Forwarded-For` value: the left-most hop is
|
||||
/// the original client (later hops are intermediary proxies). Factored out so
|
||||
/// the parsing is unit-testable without constructing a request.
|
||||
pub(crate) fn first_forwarded_ip(header: &str) -> Option<std::net::IpAddr> {
|
||||
header
|
||||
.split(',')
|
||||
.next()
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.and_then(|s| s.parse::<std::net::IpAddr>().ok())
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl FromRequestParts<AppState> for ClientIp {
|
||||
type Rejection = std::convert::Infallible;
|
||||
|
||||
async fn from_request_parts(
|
||||
parts: &mut Parts,
|
||||
state: &AppState,
|
||||
) -> Result<Self, Self::Rejection> {
|
||||
if !state.auth.trusted_proxy {
|
||||
return Ok(ClientIp(None));
|
||||
}
|
||||
let ip = parts
|
||||
.headers
|
||||
.get("x-forwarded-for")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.and_then(first_forwarded_ip);
|
||||
Ok(ClientIp(ip))
|
||||
}
|
||||
}
|
||||
|
||||
/// Cookie-only authentication. Bot/API tokens are explicitly NOT accepted
|
||||
/// here — this is the substrate for [`RequireAdmin`] and exists precisely
|
||||
/// to keep admin authority out of bearer-token reach.
|
||||
@@ -120,3 +159,34 @@ impl FromRequestParts<AppState> for RequireAdmin {
|
||||
Ok(RequireAdmin(user))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::first_forwarded_ip;
|
||||
use std::net::IpAddr;
|
||||
|
||||
#[test]
|
||||
fn parses_left_most_client_hop() {
|
||||
// The client is the first entry; later entries are intermediary proxies.
|
||||
assert_eq!(
|
||||
first_forwarded_ip("203.0.113.5, 10.0.0.1, 10.0.0.2"),
|
||||
Some("203.0.113.5".parse::<IpAddr>().unwrap())
|
||||
);
|
||||
assert_eq!(
|
||||
first_forwarded_ip(" 198.51.100.9 "),
|
||||
Some("198.51.100.9".parse::<IpAddr>().unwrap())
|
||||
);
|
||||
assert_eq!(
|
||||
first_forwarded_ip("2001:db8::1, 10.0.0.1"),
|
||||
Some("2001:db8::1".parse::<IpAddr>().unwrap())
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_empty_or_garbage() {
|
||||
assert_eq!(first_forwarded_ip(""), None);
|
||||
assert_eq!(first_forwarded_ip(" "), None);
|
||||
assert_eq!(first_forwarded_ip("not-an-ip"), None);
|
||||
assert_eq!(first_forwarded_ip(", 10.0.0.1"), None);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,6 +28,25 @@ pub fn verify_password(plain: &str, phc: &str) -> bool {
|
||||
.is_ok()
|
||||
}
|
||||
|
||||
/// Async wrapper around [`hash_password`] that offloads the CPU- and
|
||||
/// memory-heavy Argon2 work to the blocking pool. Async handlers must call
|
||||
/// this rather than the sync primitive: at ~15-50 ms per hash, running it
|
||||
/// inline stalls every other task sharing the runtime worker thread.
|
||||
pub async fn hash_password_async(plain: String) -> AppResult<String> {
|
||||
tokio::task::spawn_blocking(move || hash_password(&plain))
|
||||
.await
|
||||
.map_err(|e| AppError::Other(anyhow::anyhow!("password hash task join: {e}")))?
|
||||
}
|
||||
|
||||
/// Async wrapper around [`verify_password`]. A join failure (blocking pool
|
||||
/// gone at shutdown, panic) resolves to `false` — the caller only ever uses
|
||||
/// the result as a boolean gate, and denying auth is the safe default.
|
||||
pub async fn verify_password_async(plain: String, phc: String) -> bool {
|
||||
tokio::task::spawn_blocking(move || verify_password(&plain, &phc))
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -56,4 +75,24 @@ mod tests {
|
||||
let b = hash_password("same").unwrap();
|
||||
assert_ne!(a, b, "two hashes of the same password must differ (salt)");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_hash_then_verify_roundtrip() {
|
||||
let phc = hash_password_async("correct horse battery staple".to_string())
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(phc.starts_with("$argon2id$"));
|
||||
assert!(verify_password_async("correct horse battery staple".to_string(), phc).await);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_verify_rejects_wrong_password() {
|
||||
let phc = hash_password_async("hunter2".to_string()).await.unwrap();
|
||||
assert!(!verify_password_async("hunter3".to_string(), phc).await);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_verify_rejects_malformed_hash() {
|
||||
assert!(!verify_password_async("anything".to_string(), "not a real phc".to_string()).await);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,9 +15,17 @@
|
||||
//! tests run in isolated buckets and won't bleed across `#[sqlx::test]`
|
||||
//! cases that share a process.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::net::IpAddr;
|
||||
use std::sync::Mutex;
|
||||
use std::time::Instant;
|
||||
|
||||
/// Upper bound on distinct client IPs tracked at once, so a spray from many
|
||||
/// spoofed/rotating source IPs can't grow the map without limit. When full,
|
||||
/// idle (refilled-to-burst) buckets are pruned first; if none are idle a new
|
||||
/// IP falls back to the shared global bucket for that request.
|
||||
const MAX_TRACKED_IPS: usize = 10_000;
|
||||
|
||||
/// Tunable limits. `per_sec == 0` disables the limiter — used by the
|
||||
/// test harness and by anyone who wants to opt out via env config.
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
@@ -50,6 +58,42 @@ struct Bucket {
|
||||
last_refill: Instant,
|
||||
}
|
||||
|
||||
impl Bucket {
|
||||
fn new(burst: u32) -> Self {
|
||||
Self {
|
||||
tokens: f64::from(burst),
|
||||
last_refill: Instant::now(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Refill by elapsed time then try to consume one token.
|
||||
fn try_take(&mut self, cfg: &RateLimitConfig, now: Instant) -> AcquireResult {
|
||||
let elapsed = now.duration_since(self.last_refill).as_secs_f64();
|
||||
self.tokens =
|
||||
(self.tokens + elapsed * f64::from(cfg.per_sec)).min(f64::from(cfg.burst));
|
||||
self.last_refill = now;
|
||||
if self.tokens >= 1.0 {
|
||||
self.tokens -= 1.0;
|
||||
AcquireResult::Allowed
|
||||
} else {
|
||||
let deficit = 1.0 - self.tokens;
|
||||
let wait_secs = (deficit / f64::from(cfg.per_sec)).ceil() as u64;
|
||||
AcquireResult::Denied {
|
||||
retry_after_secs: wait_secs.max(1),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether this bucket has fully refilled (i.e. the client has been idle).
|
||||
/// Such buckets carry no state worth keeping — dropping and lazily
|
||||
/// recreating one yields an identical full bucket — so they're the safe
|
||||
/// eviction target under memory pressure.
|
||||
fn is_idle(&self, cfg: &RateLimitConfig, now: Instant) -> bool {
|
||||
let elapsed = now.duration_since(self.last_refill).as_secs_f64();
|
||||
(self.tokens + elapsed * f64::from(cfg.per_sec)) >= f64::from(cfg.burst)
|
||||
}
|
||||
}
|
||||
|
||||
/// Outcome of [`AuthRateLimiter::try_acquire`]. When `Denied`, the
|
||||
/// caller can use `retry_after_secs` for a `Retry-After: N` header
|
||||
/// (RFC 6585 §4) so well-behaved clients back off correctly rather
|
||||
@@ -60,51 +104,64 @@ pub enum AcquireResult {
|
||||
Denied { retry_after_secs: u64 },
|
||||
}
|
||||
|
||||
/// Single-bucket token-bucket limiter. `try_acquire` is cheap (one
|
||||
/// mutex acquire, no allocations) so the auth path doesn't pay a real
|
||||
/// cost for the check.
|
||||
/// Token-bucket limiter keyed by client IP, with a shared global bucket for
|
||||
/// requests whose IP is unknown (no trusted proxy).
|
||||
///
|
||||
/// The old design was a single global bucket. That let one attacker at the
|
||||
/// sustained rate deny auth to *every* user (the bucket stayed drained). With
|
||||
/// the SvelteKit proxy now forwarding the peer IP (`X-Forwarded-For`, honoured
|
||||
/// only when `AUTH_TRUSTED_PROXY` is set), each source IP gets its own bucket,
|
||||
/// so an attacker only throttles themselves. When no trustworthy IP is
|
||||
/// available the caller passes `None` and the shared `global` bucket applies —
|
||||
/// exactly the previous behaviour.
|
||||
pub struct AuthRateLimiter {
|
||||
cfg: RateLimitConfig,
|
||||
bucket: Mutex<Bucket>,
|
||||
global: Mutex<Bucket>,
|
||||
per_ip: Mutex<HashMap<IpAddr, Bucket>>,
|
||||
}
|
||||
|
||||
impl AuthRateLimiter {
|
||||
pub fn new(cfg: RateLimitConfig) -> Self {
|
||||
Self {
|
||||
cfg,
|
||||
bucket: Mutex::new(Bucket {
|
||||
tokens: cfg.burst as f64,
|
||||
last_refill: Instant::now(),
|
||||
}),
|
||||
global: Mutex::new(Bucket::new(cfg.burst)),
|
||||
per_ip: Mutex::new(HashMap::new()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Consume one token if available. Returns `Denied` with a
|
||||
/// rounded-up seconds-until-refill so the caller can emit a
|
||||
/// `Retry-After` header.
|
||||
pub fn try_acquire(&self) -> AcquireResult {
|
||||
/// Consume one token from the bucket for `key` (per-IP when `Some`, the
|
||||
/// shared global bucket when `None`). Returns `Denied` with a rounded-up
|
||||
/// seconds-until-refill so the caller can emit a `Retry-After` header.
|
||||
pub fn try_acquire(&self, key: Option<IpAddr>) -> AcquireResult {
|
||||
if self.cfg.per_sec == 0 {
|
||||
return AcquireResult::Allowed;
|
||||
}
|
||||
let now = Instant::now();
|
||||
let mut bucket = self.bucket.lock().expect("rate limiter mutex poisoned");
|
||||
let elapsed = now.duration_since(bucket.last_refill).as_secs_f64();
|
||||
bucket.tokens =
|
||||
(bucket.tokens + elapsed * f64::from(self.cfg.per_sec)).min(f64::from(self.cfg.burst));
|
||||
bucket.last_refill = now;
|
||||
if bucket.tokens >= 1.0 {
|
||||
bucket.tokens -= 1.0;
|
||||
AcquireResult::Allowed
|
||||
} else {
|
||||
// ceil((1 - tokens) / per_sec), minimum 1 — a `Retry-After: 0`
|
||||
// would tell clients to retry immediately, which is what we're
|
||||
// actively trying to discourage.
|
||||
let deficit = 1.0 - bucket.tokens;
|
||||
let wait_secs = (deficit / f64::from(self.cfg.per_sec)).ceil() as u64;
|
||||
AcquireResult::Denied {
|
||||
retry_after_secs: wait_secs.max(1),
|
||||
let Some(ip) = key else {
|
||||
return self
|
||||
.global
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.try_take(&self.cfg, now);
|
||||
};
|
||||
let mut map = self.per_ip.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if map.len() >= MAX_TRACKED_IPS && !map.contains_key(&ip) {
|
||||
map.retain(|_, b| !b.is_idle(&self.cfg, now));
|
||||
if map.len() >= MAX_TRACKED_IPS {
|
||||
// Still saturated with active attackers — degrade to the shared
|
||||
// bucket rather than grow unbounded. Worst case is the old
|
||||
// global-bucket behaviour under an extreme distributed flood.
|
||||
drop(map);
|
||||
return self
|
||||
.global
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.try_take(&self.cfg, now);
|
||||
}
|
||||
}
|
||||
map.entry(ip)
|
||||
.or_insert_with(|| Bucket::new(self.cfg.burst))
|
||||
.try_take(&self.cfg, now)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -119,7 +176,7 @@ mod tests {
|
||||
burst: 0,
|
||||
});
|
||||
for _ in 0..1000 {
|
||||
assert_eq!(rl.try_acquire(), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -130,10 +187,10 @@ mod tests {
|
||||
per_sec: 1,
|
||||
burst: 3,
|
||||
});
|
||||
assert_eq!(rl.try_acquire(), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(), AcquireResult::Allowed);
|
||||
match rl.try_acquire() {
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
match rl.try_acquire(None) {
|
||||
AcquireResult::Denied { retry_after_secs } => {
|
||||
// Bucket is at ~0 tokens, refill rate 1/sec → ~1s wait.
|
||||
assert!(
|
||||
@@ -152,11 +209,11 @@ mod tests {
|
||||
per_sec: 10,
|
||||
burst: 1,
|
||||
});
|
||||
assert_eq!(rl.try_acquire(), AcquireResult::Allowed);
|
||||
assert!(matches!(rl.try_acquire(), AcquireResult::Denied { .. }));
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
assert!(matches!(rl.try_acquire(None), AcquireResult::Denied { .. }));
|
||||
std::thread::sleep(std::time::Duration::from_millis(150));
|
||||
assert_eq!(
|
||||
rl.try_acquire(),
|
||||
rl.try_acquire(None),
|
||||
AcquireResult::Allowed,
|
||||
"token should have refilled"
|
||||
);
|
||||
@@ -170,10 +227,95 @@ mod tests {
|
||||
per_sec: 1,
|
||||
burst: 1,
|
||||
});
|
||||
slow.try_acquire();
|
||||
match slow.try_acquire() {
|
||||
slow.try_acquire(None);
|
||||
match slow.try_acquire(None) {
|
||||
AcquireResult::Denied { retry_after_secs } => assert_eq!(retry_after_secs, 1),
|
||||
_ => panic!("expected Denied"),
|
||||
}
|
||||
}
|
||||
|
||||
fn ip(s: &str) -> Option<IpAddr> {
|
||||
Some(s.parse().unwrap())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn per_ip_buckets_are_independent() {
|
||||
// The whole point of the fix: one IP draining its bucket must not deny
|
||||
// a different IP.
|
||||
let rl = AuthRateLimiter::new(RateLimitConfig {
|
||||
per_sec: 1,
|
||||
burst: 2,
|
||||
});
|
||||
let a = ip("203.0.113.7");
|
||||
let b = ip("198.51.100.9");
|
||||
assert_eq!(rl.try_acquire(a), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(a), AcquireResult::Allowed);
|
||||
assert!(matches!(rl.try_acquire(a), AcquireResult::Denied { .. }));
|
||||
// b is untouched.
|
||||
assert_eq!(rl.try_acquire(b), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(b), AcquireResult::Allowed);
|
||||
assert!(matches!(rl.try_acquire(b), AcquireResult::Denied { .. }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn none_key_shares_the_global_bucket_independent_of_per_ip() {
|
||||
let rl = AuthRateLimiter::new(RateLimitConfig {
|
||||
per_sec: 1,
|
||||
burst: 2,
|
||||
});
|
||||
// Drain the global (None) bucket.
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
assert!(matches!(rl.try_acquire(None), AcquireResult::Denied { .. }));
|
||||
// A real IP is on its own bucket, unaffected by the drained global one.
|
||||
assert_eq!(rl.try_acquire(ip("203.0.113.1")), AcquireResult::Allowed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tracked_ip_map_is_bounded() {
|
||||
let rl = AuthRateLimiter::new(RateLimitConfig {
|
||||
per_sec: 1,
|
||||
burst: 1,
|
||||
});
|
||||
// Distinct IPs beyond the cap must not grow the map without bound —
|
||||
// excess requests fall back to the shared bucket instead.
|
||||
for i in 0..(MAX_TRACKED_IPS as u64 + 50) {
|
||||
let octet_a = (i >> 8) as u8;
|
||||
let octet_b = (i & 0xff) as u8;
|
||||
let addr = format!("10.20.{octet_a}.{octet_b}");
|
||||
let _ = rl.try_acquire(Some(addr.parse().unwrap()));
|
||||
}
|
||||
assert!(rl.per_ip.lock().unwrap().len() <= MAX_TRACKED_IPS);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn survives_a_poisoned_mutex() {
|
||||
// If any thread ever panics while holding a limiter mutex, the lock
|
||||
// becomes poisoned. With the old `.expect(...)` every later auth request
|
||||
// would re-panic — one blip turned into a permanent auth outage. Recover
|
||||
// the guard via `into_inner()` instead so the limiter keeps serving.
|
||||
let rl = AuthRateLimiter::new(RateLimitConfig {
|
||||
per_sec: 5,
|
||||
burst: 5,
|
||||
});
|
||||
|
||||
// Poison per_ip by panicking while holding its guard.
|
||||
let poisoned = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
let _g = rl.per_ip.lock().unwrap();
|
||||
panic!("poison the per-IP mutex");
|
||||
}));
|
||||
assert!(poisoned.is_err(), "the panic must unwind");
|
||||
assert!(rl.per_ip.is_poisoned(), "the mutex must now be poisoned");
|
||||
|
||||
// The per-IP path (Some(ip)) must still work despite the poison.
|
||||
assert_eq!(rl.try_acquire(ip("198.51.100.9")), AcquireResult::Allowed);
|
||||
|
||||
// And poison the global bucket too — the None-key path must recover.
|
||||
let _ = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
let _g = rl.global.lock().unwrap();
|
||||
panic!("poison the global mutex");
|
||||
}));
|
||||
assert!(rl.global.is_poisoned());
|
||||
assert_eq!(rl.try_acquire(None), AcquireResult::Allowed);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,15 +3,16 @@
|
||||
//! `generate_token` draws 32 bytes from the OS CSPRNG, encodes them as
|
||||
//! URL-safe base64 (no padding), and returns the raw string alongside its
|
||||
//! SHA-256 hash. Storage holds only the hash; the raw value lives in the
|
||||
//! cookie or `Authorization` header. Comparison goes through
|
||||
//! `constant_time_eq` to keep timing side channels off the table.
|
||||
//! cookie or `Authorization` header. Token lookup is an indexed equality on
|
||||
//! that 256-bit hash in the database (`WHERE token_hash = $1`), so there's no
|
||||
//! in-process secret comparison to time-attack: a guess has to match a full
|
||||
//! SHA-256 digest, and the DB index reveals nothing about how close it came.
|
||||
|
||||
use base64::engine::general_purpose::URL_SAFE_NO_PAD;
|
||||
use base64::Engine as _;
|
||||
use rand::rngs::OsRng;
|
||||
use rand::RngCore;
|
||||
use sha2::{Digest, Sha256};
|
||||
use subtle::ConstantTimeEq;
|
||||
|
||||
pub const TOKEN_BYTES: usize = 32;
|
||||
pub const HASH_BYTES: usize = 32;
|
||||
@@ -30,10 +31,6 @@ pub fn hash_token(raw: &str) -> [u8; HASH_BYTES] {
|
||||
hasher.finalize().into()
|
||||
}
|
||||
|
||||
pub fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
|
||||
a.ct_eq(b).into()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -58,11 +55,4 @@ mod tests {
|
||||
assert_eq!(hash_token("abc"), hash_token("abc"));
|
||||
assert_ne!(hash_token("abc"), hash_token("abd"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_time_eq_compares_correctly() {
|
||||
assert!(constant_time_eq(b"abc", b"abc"));
|
||||
assert!(!constant_time_eq(b"abc", b"abd"));
|
||||
assert!(!constant_time_eq(b"abc", b"abcd"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,9 +112,20 @@ async fn main() -> anyhow::Result<()> {
|
||||
cookie_jar.add_cookie_str(&cookie_str, &seed_url);
|
||||
tracing::info!(domain, "seeded PHPSESSID into reqwest cookie jar");
|
||||
}
|
||||
// SSRF defence: only download from the catalog host + CDN host (plus
|
||||
// optional CRAWLER_DOWNLOAD_ALLOWLIST extras). Built here so the same
|
||||
// allowlist guards both the redirect policy and the per-image check.
|
||||
let allowlist = Arc::new(build_download_allowlist(&start_url, cdn_host.as_deref()));
|
||||
let mut http_builder = reqwest::Client::builder()
|
||||
.timeout(Duration::from_secs(30))
|
||||
.no_proxy()
|
||||
// Re-validate every redirect hop against the allowlist — reqwest's
|
||||
// default follows up to 10 redirects and `is_safe_url` only guards
|
||||
// the initial URL, so a 302 to a private IP would otherwise pivot
|
||||
// inside the deployment (SSRF).
|
||||
.redirect(mangalord::crawler::safety::safe_redirect_policy(
|
||||
(*allowlist).clone(),
|
||||
))
|
||||
.cookie_provider(cookie_jar);
|
||||
if let Some(ua) = &user_agent {
|
||||
http_builder = http_builder.user_agent(ua);
|
||||
@@ -123,8 +134,25 @@ async fn main() -> anyhow::Result<()> {
|
||||
http_builder = http_builder
|
||||
.proxy(reqwest::Proxy::all(proxy).with_context(|| format!("parse proxy URL: {proxy}"))?);
|
||||
}
|
||||
// DNS-rebinding guard: attached on the direct path AND on http(s) proxies
|
||||
// (reqwest resolves the target itself there), skipped only for SOCKS proxies
|
||||
// where the proxy resolves the target. Use the shared predicate so this CLI
|
||||
// and the daemon (app.rs) stay in lockstep — previously the CLI attached the
|
||||
// resolver only on the fully-direct path, so an http(s) proxy silently lost
|
||||
// the rebinding guard.
|
||||
if mangalord::crawler::safety::should_attach_safe_resolver(proxy_url.as_deref()) {
|
||||
http_builder =
|
||||
http_builder.dns_resolver(mangalord::crawler::safety::safe_dns_resolver());
|
||||
}
|
||||
let http = http_builder.build().context("build http client")?;
|
||||
|
||||
// Opt-in browser SSRF interception (default off), mirroring the daemon.
|
||||
mangalord::crawler::intercept::set_enabled(
|
||||
std::env::var("CRAWLER_SSRF_INTERCEPT")
|
||||
.map(|v| matches!(v.as_str(), "1" | "true" | "TRUE" | "yes"))
|
||||
.unwrap_or(false),
|
||||
);
|
||||
|
||||
let mut options = LaunchOptions::from_env();
|
||||
if let Some(proxy) = &proxy_url {
|
||||
let chromium_proxy = mangalord::crawler::url_utils::chromium_proxy_arg(proxy);
|
||||
@@ -216,6 +244,7 @@ async fn main() -> anyhow::Result<()> {
|
||||
rate_ms,
|
||||
cdn_host.as_deref(),
|
||||
cdn_rate_ms,
|
||||
Arc::clone(&allowlist),
|
||||
limit,
|
||||
skip_chapters,
|
||||
skip_chapter_content || !session_ready,
|
||||
@@ -246,6 +275,7 @@ async fn run(
|
||||
rate_ms: u64,
|
||||
cdn_host: Option<&str>,
|
||||
cdn_rate_ms: u64,
|
||||
allowlist: Arc<mangalord::crawler::safety::DownloadAllowlist>,
|
||||
limit: usize,
|
||||
skip_chapters: bool,
|
||||
skip_chapter_content: bool,
|
||||
@@ -259,38 +289,18 @@ async fn run(
|
||||
}
|
||||
let rate = Arc::new(rate);
|
||||
|
||||
// SSRF defence: only download from the catalog host + CDN host
|
||||
// (plus optional CRAWLER_DOWNLOAD_ALLOWLIST extras), and cap
|
||||
// single-image downloads at CRAWLER_MAX_IMAGE_BYTES bytes.
|
||||
// CRAWLER_ALLOW_ANY_HOST=true short-circuits the host check for
|
||||
// sharded-CDN sources; private-IP and scheme guards still apply.
|
||||
let allowlist = if env_bool("CRAWLER_ALLOW_ANY_HOST", false) {
|
||||
mangalord::crawler::safety::DownloadAllowlist::allow_any()
|
||||
} else {
|
||||
let mut allow = mangalord::crawler::safety::DownloadAllowlist::new();
|
||||
if let Ok(parsed) = reqwest::Url::parse(start_url) {
|
||||
if let Some(h) = parsed.host_str() {
|
||||
allow = allow.allow(h);
|
||||
}
|
||||
}
|
||||
if let Some(host) = cdn_host {
|
||||
allow = allow.allow(host);
|
||||
}
|
||||
if let Ok(extras) = std::env::var("CRAWLER_DOWNLOAD_ALLOWLIST") {
|
||||
for piece in extras.split(',') {
|
||||
let trimmed = piece.trim();
|
||||
if !trimmed.is_empty() {
|
||||
allow = allow.allow(trimmed);
|
||||
}
|
||||
}
|
||||
}
|
||||
allow
|
||||
};
|
||||
// Per-image download cap (the allowlist is built in `main` and passed in
|
||||
// so the HTTP client's redirect policy and this check share one source).
|
||||
let max_image_bytes: usize = std::env::var("CRAWLER_MAX_IMAGE_BYTES")
|
||||
.ok()
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or(mangalord::crawler::safety::DEFAULT_MAX_IMAGE_BYTES);
|
||||
let allowlist = Arc::new(allowlist);
|
||||
// Per-chapter image *count* cap — bounds total disk against a hostile
|
||||
// reader page listing thousands of <img> tags. `0` disables it.
|
||||
let max_images_per_chapter: usize = std::env::var("CRAWLER_MAX_IMAGES_PER_CHAPTER")
|
||||
.ok()
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or(2000);
|
||||
|
||||
let stats = pipeline::run_metadata_pass(
|
||||
manager.as_ref(),
|
||||
@@ -325,6 +335,7 @@ async fn run(
|
||||
force_refetch_chapters,
|
||||
Arc::clone(&allowlist),
|
||||
max_image_bytes,
|
||||
max_images_per_chapter,
|
||||
tor.clone(),
|
||||
)
|
||||
.await?;
|
||||
@@ -351,6 +362,7 @@ async fn sync_bookmarked_chapter_content(
|
||||
force_refetch: bool,
|
||||
allowlist: Arc<mangalord::crawler::safety::DownloadAllowlist>,
|
||||
max_image_bytes: usize,
|
||||
max_images_per_chapter: usize,
|
||||
tor: Option<Arc<mangalord::crawler::tor::TorController>>,
|
||||
) -> anyhow::Result<()> {
|
||||
let pending: Vec<(Uuid, Uuid, String)> = sqlx::query_as(
|
||||
@@ -416,6 +428,7 @@ async fn sync_bookmarked_chapter_content(
|
||||
force_refetch,
|
||||
allowlist.as_ref(),
|
||||
max_image_bytes,
|
||||
max_images_per_chapter,
|
||||
tor.as_deref(),
|
||||
// CLI one-shot — no live status surface.
|
||||
None,
|
||||
@@ -431,6 +444,13 @@ async fn sync_bookmarked_chapter_content(
|
||||
s.fetched += 1;
|
||||
}
|
||||
Ok(SyncOutcome::Skipped) => s.skipped += 1,
|
||||
// Unreachable in the one-shot CLI (it holds its own lease
|
||||
// and never dispatches through the queue), but count it as a
|
||||
// failure for exhaustiveness.
|
||||
Ok(SyncOutcome::BrowserUnavailable) => {
|
||||
tracing::warn!(%chapter_id, "crawler browser unavailable");
|
||||
s.failed += 1;
|
||||
}
|
||||
Ok(SyncOutcome::SessionExpired) => {
|
||||
tracing::error!(
|
||||
%chapter_id,
|
||||
@@ -497,3 +517,37 @@ fn env_bool(name: &str, default: bool) -> bool {
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the crawler download allowlist from env + the catalog/CDN hosts.
|
||||
/// Shared by the HTTP client's redirect policy and the per-image safety
|
||||
/// check so both agree on which hosts are reachable.
|
||||
///
|
||||
/// `CRAWLER_ALLOW_ANY_HOST=true` short-circuits the host check for
|
||||
/// sharded-CDN sources; private-IP and scheme guards still apply.
|
||||
fn build_download_allowlist(
|
||||
start_url: &str,
|
||||
cdn_host: Option<&str>,
|
||||
) -> mangalord::crawler::safety::DownloadAllowlist {
|
||||
use mangalord::crawler::safety::DownloadAllowlist;
|
||||
if env_bool("CRAWLER_ALLOW_ANY_HOST", false) {
|
||||
return DownloadAllowlist::allow_any();
|
||||
}
|
||||
let mut allow = DownloadAllowlist::new();
|
||||
if let Ok(parsed) = reqwest::Url::parse(start_url) {
|
||||
if let Some(h) = parsed.host_str() {
|
||||
allow = allow.allow(h);
|
||||
}
|
||||
}
|
||||
if let Some(host) = cdn_host {
|
||||
allow = allow.allow(host);
|
||||
}
|
||||
if let Ok(extras) = std::env::var("CRAWLER_DOWNLOAD_ALLOWLIST") {
|
||||
for piece in extras.split(',') {
|
||||
let trimmed = piece.trim();
|
||||
if !trimmed.is_empty() {
|
||||
allow = allow.allow(trimmed);
|
||||
}
|
||||
}
|
||||
}
|
||||
allow
|
||||
}
|
||||
|
||||
|
||||
@@ -27,6 +27,14 @@ pub struct AuthConfig {
|
||||
/// so a private instance is locked down with a single switch.
|
||||
/// Defaults to `false` (current public behaviour).
|
||||
pub private_mode: bool,
|
||||
/// Whether to trust a proxy-supplied `X-Forwarded-For` header as the
|
||||
/// client IP for per-IP auth rate limiting. Enable ONLY when the backend
|
||||
/// sits behind a trusted reverse proxy that overrides the header (the
|
||||
/// compose deploy: SvelteKit's hooks.server.ts sets it from the real peer
|
||||
/// address). When `false` (default), the header is ignored and the auth
|
||||
/// limiter uses a single shared bucket — a directly-exposed backend must
|
||||
/// keep this off or clients could spoof their IP to dodge the limit.
|
||||
pub trusted_proxy: bool,
|
||||
}
|
||||
|
||||
impl Default for AuthConfig {
|
||||
@@ -42,6 +50,7 @@ impl Default for AuthConfig {
|
||||
rate_limit: crate::auth::rate_limit::RateLimitConfig::default(),
|
||||
allow_self_register: true,
|
||||
private_mode: false,
|
||||
trusted_proxy: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -55,6 +64,13 @@ pub struct UploadConfig {
|
||||
/// reject a single oversized cover/page without failing the whole
|
||||
/// request just because the total happens to fit.
|
||||
pub max_file_bytes: usize,
|
||||
/// Max page images accepted in one chapter upload. Bounds how many
|
||||
/// parts the handler will stage before giving up, so a client can't
|
||||
/// pin a worker streaming an unbounded page count. `0` disables THIS
|
||||
/// cap — the total is then bounded only by `max_request_bytes` (the
|
||||
/// whole-request body limit), which stays the backstop either way.
|
||||
/// Defaults to 2000. `MAX_PAGES_PER_CHAPTER`.
|
||||
pub max_pages_per_chapter: usize,
|
||||
}
|
||||
|
||||
impl Default for UploadConfig {
|
||||
@@ -62,6 +78,46 @@ impl Default for UploadConfig {
|
||||
Self {
|
||||
max_request_bytes: 200 * 1024 * 1024, // 200 MiB
|
||||
max_file_bytes: 20 * 1024 * 1024, // 20 MiB
|
||||
max_pages_per_chapter: 2000,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Postgres connection-pool sizing. One pool backs every HTTP handler plus
|
||||
/// the crawler/analysis daemons, so it must be large enough not to starve
|
||||
/// interactive reads and fail fast (rather than hang on the driver's silent
|
||||
/// 30 s default) when genuinely saturated.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct DbConfig {
|
||||
/// `DB_MAX_CONNECTIONS`. Upper bound on open connections.
|
||||
pub max_connections: u32,
|
||||
/// `DB_ACQUIRE_TIMEOUT_SECS`. How long a caller waits for a free
|
||||
/// connection before erroring — short so overload surfaces as a fast
|
||||
/// 500 instead of a 30 s hang.
|
||||
pub acquire_timeout: Duration,
|
||||
}
|
||||
|
||||
impl Default for DbConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
max_connections: 20,
|
||||
acquire_timeout: Duration::from_secs(10),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl DbConfig {
|
||||
pub fn from_env() -> Self {
|
||||
let default = Self::default();
|
||||
Self {
|
||||
// `.max(1)`: a zero-size pool can never hand out a connection and
|
||||
// would deadlock every query — clamp to at least one.
|
||||
max_connections: env_u64("DB_MAX_CONNECTIONS", default.max_connections.into())
|
||||
.max(1) as u32,
|
||||
acquire_timeout: Duration::from_secs(env_u64(
|
||||
"DB_ACQUIRE_TIMEOUT_SECS",
|
||||
default.acquire_timeout.as_secs(),
|
||||
)),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -113,6 +169,32 @@ impl ResponseFormat {
|
||||
}
|
||||
}
|
||||
|
||||
/// Which engine the analysis worker dispatches each page through. A
|
||||
/// deploy-time choice (the engine is either installed or not), so it lives in
|
||||
/// env only and is not part of the admin-editable `AnalysisSettings`.
|
||||
///
|
||||
/// * `Ocr` — in-process [`crate::analysis::ocr`] (the `ocrs` engine): fast,
|
||||
/// CPU-only, English text. Writes OCR text only (no tags/scene/safety).
|
||||
/// * `Vision` — the local OpenAI-compatible LLM in [`crate::analysis::vision`]:
|
||||
/// full OCR + tags + scene + safety, but heavy.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum AnalysisBackend {
|
||||
Ocr,
|
||||
Vision,
|
||||
}
|
||||
|
||||
impl AnalysisBackend {
|
||||
/// Lenient env parse. Defaults to `Ocr` (the Pi-friendly path) for any
|
||||
/// unset or unrecognized value; `vision` opts back into the LLM engine.
|
||||
fn from_str(s: &str) -> AnalysisBackend {
|
||||
match s.trim().to_lowercase().as_str() {
|
||||
"vision" | "llm" => AnalysisBackend::Vision,
|
||||
// Default (incl. "ocr", "ocrs", and anything unrecognized).
|
||||
_ => AnalysisBackend::Ocr,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// AI content-analysis worker configuration: the enable gate, the local
|
||||
/// OpenAI-compatible vision endpoint, and the worker / request knobs.
|
||||
#[derive(Clone, Debug)]
|
||||
@@ -120,6 +202,16 @@ pub struct AnalysisConfig {
|
||||
/// Master switch (`ANALYSIS_ENABLED`). When `false`, no analysis jobs
|
||||
/// are enqueued and no worker runs. Defaults to `false`.
|
||||
pub enabled: bool,
|
||||
/// Which engine the worker dispatches through (`ANALYSIS_BACKEND`):
|
||||
/// `ocr` (default, the in-process `ocrs` engine) or `vision` (the local
|
||||
/// LLM). Deploy-time, env-only — see [`AnalysisBackend`].
|
||||
pub backend: AnalysisBackend,
|
||||
/// Path to the ocrs text-*detection* `.rten` model
|
||||
/// (`OCRS_DETECTION_MODEL`). Only read when `backend == Ocr`.
|
||||
pub ocr_detection_model: String,
|
||||
/// Path to the ocrs text-*recognition* `.rten` model
|
||||
/// (`OCRS_RECOGNITION_MODEL`). Only read when `backend == Ocr`.
|
||||
pub ocr_recognition_model: String,
|
||||
/// Number of concurrent analysis workers (`ANALYSIS_WORKERS`).
|
||||
pub workers: usize,
|
||||
/// OpenAI-compatible chat/completions URL (`ANALYSIS_VISION_URL`).
|
||||
@@ -166,6 +258,13 @@ pub struct AnalysisConfig {
|
||||
/// Hard cap on a page image's stored size; larger pages are skipped
|
||||
/// (`ANALYSIS_MAX_IMAGE_BYTES`).
|
||||
pub max_image_bytes: usize,
|
||||
/// Hard cap on a page image's **decoded** pixel count for the OCR backend
|
||||
/// (`ANALYSIS_OCR_MAX_DECODE_PIXELS`). `max_image_bytes` only bounds the
|
||||
/// *encoded* size; without a decode bound a tiny image declaring
|
||||
/// 50000×50000 inflates to billions of bytes and OOM-kills the worker
|
||||
/// (decompression bomb). Generous by default (100 MP) so legitimately
|
||||
/// tall, un-sliced manga pages still decode.
|
||||
pub ocr_max_decode_pixels: u64,
|
||||
/// Output-constraint mode (`ANALYSIS_RESPONSE_FORMAT`):
|
||||
/// `json_schema` (default) | `json_object` | `none`.
|
||||
pub response_format: ResponseFormat,
|
||||
@@ -195,6 +294,9 @@ impl Default for AnalysisConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
enabled: false,
|
||||
backend: AnalysisBackend::Ocr,
|
||||
ocr_detection_model: "/models/text-detection.rten".to_string(),
|
||||
ocr_recognition_model: "/models/text-recognition.rten".to_string(),
|
||||
workers: 1,
|
||||
endpoint: "http://localhost:8000/v1/chat/completions".to_string(),
|
||||
vision_health_url: None,
|
||||
@@ -212,6 +314,7 @@ impl Default for AnalysisConfig {
|
||||
tall_aspect_threshold: 1.6,
|
||||
max_slices: 16,
|
||||
max_image_bytes: 8 * 1024 * 1024,
|
||||
ocr_max_decode_pixels: 100_000_000,
|
||||
response_format: ResponseFormat::JsonSchema,
|
||||
frequency_penalty: 0.3,
|
||||
temperature: 0.0,
|
||||
@@ -223,10 +326,39 @@ impl Default for AnalysisConfig {
|
||||
}
|
||||
|
||||
impl AnalysisConfig {
|
||||
/// The backend the worker actually dispatches through.
|
||||
///
|
||||
/// Vision is **temporarily disabled**: the engine code (`analysis::vision`,
|
||||
/// `RealAnalyzeDispatcher`, the readiness probe) is kept intact but never
|
||||
/// selected. Until it's re-enabled, this returns [`AnalysisBackend::Ocr`]
|
||||
/// regardless of the parsed `backend`, logging a warning if `vision` was
|
||||
/// requested so an env override isn't silently ignored. Re-enabling vision
|
||||
/// is then a one-line change here (return `self.backend`).
|
||||
pub fn effective_backend(&self) -> AnalysisBackend {
|
||||
if self.backend == AnalysisBackend::Vision {
|
||||
tracing::warn!(
|
||||
"ANALYSIS_BACKEND=vision requested but the vision backend is temporarily \
|
||||
disabled; running OCR instead"
|
||||
);
|
||||
}
|
||||
AnalysisBackend::Ocr
|
||||
}
|
||||
|
||||
pub fn from_env() -> Self {
|
||||
let d = AnalysisConfig::default();
|
||||
Self {
|
||||
enabled: env_bool("ANALYSIS_ENABLED", d.enabled),
|
||||
backend: std::env::var("ANALYSIS_BACKEND")
|
||||
.map(|s| AnalysisBackend::from_str(&s))
|
||||
.unwrap_or(d.backend),
|
||||
ocr_detection_model: std::env::var("OCRS_DETECTION_MODEL")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or(d.ocr_detection_model),
|
||||
ocr_recognition_model: std::env::var("OCRS_RECOGNITION_MODEL")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or(d.ocr_recognition_model),
|
||||
workers: env_usize("ANALYSIS_WORKERS", d.workers).max(1),
|
||||
endpoint: std::env::var("ANALYSIS_VISION_URL").unwrap_or(d.endpoint),
|
||||
vision_health_url: std::env::var("ANALYSIS_VISION_HEALTH_URL")
|
||||
@@ -259,6 +391,10 @@ impl AnalysisConfig {
|
||||
.max(1.0),
|
||||
max_slices: env_usize("ANALYSIS_MAX_SLICES", d.max_slices).max(1),
|
||||
max_image_bytes: env_usize("ANALYSIS_MAX_IMAGE_BYTES", d.max_image_bytes),
|
||||
ocr_max_decode_pixels: env_u64(
|
||||
"ANALYSIS_OCR_MAX_DECODE_PIXELS",
|
||||
d.ocr_max_decode_pixels,
|
||||
),
|
||||
response_format: std::env::var("ANALYSIS_RESPONSE_FORMAT")
|
||||
.map(|s| ResponseFormat::from_str(&s))
|
||||
.unwrap_or(d.response_format),
|
||||
@@ -285,6 +421,7 @@ pub struct Config {
|
||||
pub database_url: String,
|
||||
pub bind_address: String,
|
||||
pub storage_dir: PathBuf,
|
||||
pub db: DbConfig,
|
||||
pub auth: AuthConfig,
|
||||
pub upload: UploadConfig,
|
||||
pub cors_allowed_origins: Vec<String>,
|
||||
@@ -361,6 +498,13 @@ pub struct CrawlerConfig {
|
||||
pub download_allowlist: DownloadAllowlist,
|
||||
/// Hard upper bound on a single image download. Defaults to 32 MiB.
|
||||
pub max_image_bytes: usize,
|
||||
/// Hard upper bound on the number of page images in one chapter. A
|
||||
/// hostile reader page could otherwise list thousands of `<img>`
|
||||
/// tags; `max_image_bytes` caps each one but not the count, so the
|
||||
/// product is an unbounded disk-fill. A chapter exceeding this is
|
||||
/// acked failed rather than downloaded. `0` disables the cap.
|
||||
/// Defaults to 2000. `CRAWLER_MAX_IMAGES_PER_CHAPTER`.
|
||||
pub max_images_per_chapter: usize,
|
||||
/// Max manga detail fetches per metadata pass. `0` means no cap
|
||||
/// (full sweep up to the source's own bound). Sourced from
|
||||
/// `CRAWLER_LIMIT`, mirroring the CLI binary.
|
||||
@@ -378,6 +522,16 @@ pub struct CrawlerConfig {
|
||||
/// exhausted) that trigger an automatic coordinated browser restart.
|
||||
/// Defaults to 3. `CRAWLER_BROWSER_RESTART_THRESHOLD`.
|
||||
pub browser_restart_threshold: u32,
|
||||
/// CDP `Fetch` interception that re-validates every headless-browser
|
||||
/// navigation/redirect/subresource against the SSRF check. Default `true`:
|
||||
/// with it off, only the top-level URL string is validated, so a scraped
|
||||
/// page's JS/subresources (which use Chromium's own network stack, not the
|
||||
/// reqwest `SafeResolver`) can reach internal targets like the cloud
|
||||
/// metadata service or postgres. The Fetch hook can't be exercised in CI (no
|
||||
/// Chromium) — the `#[ignore]`d `ssrf_interception_does_not_wedge_allowed_navigation`
|
||||
/// smoke test validates it against a real binary. `CRAWLER_SSRF_INTERCEPT`;
|
||||
/// set `false` only as a break-glass if the hook destabilizes a deployment.
|
||||
pub ssrf_intercept: bool,
|
||||
}
|
||||
|
||||
impl Default for CrawlerConfig {
|
||||
@@ -405,10 +559,12 @@ impl Default for CrawlerConfig {
|
||||
browser: LaunchOptions::headless(),
|
||||
download_allowlist: DownloadAllowlist::new(),
|
||||
max_image_bytes: DEFAULT_MAX_IMAGE_BYTES,
|
||||
max_images_per_chapter: 2000,
|
||||
manga_limit: 0,
|
||||
job_timeout: Duration::from_secs(600),
|
||||
metadata_max_consecutive_failures: 10,
|
||||
browser_restart_threshold: 3,
|
||||
ssrf_intercept: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -423,6 +579,7 @@ impl Config {
|
||||
storage_dir: std::env::var("STORAGE_DIR")
|
||||
.unwrap_or_else(|_| "./data/storage".to_string())
|
||||
.into(),
|
||||
db: DbConfig::from_env(),
|
||||
auth: AuthConfig {
|
||||
cookie_secure: env_bool("COOKIE_SECURE", true),
|
||||
cookie_domain: std::env::var("COOKIE_DOMAIN")
|
||||
@@ -441,10 +598,12 @@ impl Config {
|
||||
},
|
||||
allow_self_register: env_bool("ALLOW_SELF_REGISTER", true),
|
||||
private_mode: env_bool("PRIVATE_MODE", false),
|
||||
trusted_proxy: env_bool("AUTH_TRUSTED_PROXY", false),
|
||||
},
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: env_usize("MAX_REQUEST_BYTES", 200 * 1024 * 1024),
|
||||
max_file_bytes: env_usize("MAX_FILE_BYTES", 20 * 1024 * 1024),
|
||||
max_pages_per_chapter: env_usize("MAX_PAGES_PER_CHAPTER", 2000),
|
||||
},
|
||||
cors_allowed_origins: std::env::var("CORS_ALLOWED_ORIGINS")
|
||||
.ok()
|
||||
@@ -562,6 +721,7 @@ impl CrawlerConfig {
|
||||
browser: LaunchOptions::from_env(),
|
||||
download_allowlist,
|
||||
max_image_bytes: env_usize("CRAWLER_MAX_IMAGE_BYTES", DEFAULT_MAX_IMAGE_BYTES),
|
||||
max_images_per_chapter: env_usize("CRAWLER_MAX_IMAGES_PER_CHAPTER", 2000),
|
||||
manga_limit: env_usize("CRAWLER_LIMIT", 0),
|
||||
job_timeout: Duration::from_secs(env_u64("CRAWLER_JOB_TIMEOUT_SECS", 600).max(1)),
|
||||
metadata_max_consecutive_failures: env_u64(
|
||||
@@ -570,6 +730,7 @@ impl CrawlerConfig {
|
||||
) as u32,
|
||||
browser_restart_threshold: env_u64("CRAWLER_BROWSER_RESTART_THRESHOLD", 3).max(1)
|
||||
as u32,
|
||||
ssrf_intercept: env_bool("CRAWLER_SSRF_INTERCEPT", true),
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -717,6 +878,37 @@ mod tests {
|
||||
assert_eq!(cfg.browser_restart_threshold, 7);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn db_pool_defaults_when_unset() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
std::env::remove_var("DB_MAX_CONNECTIONS");
|
||||
std::env::remove_var("DB_ACQUIRE_TIMEOUT_SECS");
|
||||
let cfg = DbConfig::from_env();
|
||||
assert_eq!(cfg.max_connections, 20);
|
||||
assert_eq!(cfg.acquire_timeout, Duration::from_secs(10));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn db_pool_parses_from_env() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
std::env::set_var("DB_MAX_CONNECTIONS", "50");
|
||||
std::env::set_var("DB_ACQUIRE_TIMEOUT_SECS", "3");
|
||||
let cfg = DbConfig::from_env();
|
||||
std::env::remove_var("DB_MAX_CONNECTIONS");
|
||||
std::env::remove_var("DB_ACQUIRE_TIMEOUT_SECS");
|
||||
assert_eq!(cfg.max_connections, 50);
|
||||
assert_eq!(cfg.acquire_timeout, Duration::from_secs(3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn db_pool_max_connections_clamps_to_at_least_one() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
std::env::set_var("DB_MAX_CONNECTIONS", "0");
|
||||
let cfg = DbConfig::from_env();
|
||||
std::env::remove_var("DB_MAX_CONNECTIONS");
|
||||
assert_eq!(cfg.max_connections, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_config_defaults_when_unset() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
@@ -734,11 +926,18 @@ mod tests {
|
||||
"ANALYSIS_MAX_SLICES",
|
||||
"ANALYSIS_RESPONSE_FORMAT",
|
||||
"ANALYSIS_FREQUENCY_PENALTY",
|
||||
"ANALYSIS_BACKEND",
|
||||
"OCRS_DETECTION_MODEL",
|
||||
"OCRS_RECOGNITION_MODEL",
|
||||
] {
|
||||
std::env::remove_var(k);
|
||||
}
|
||||
let cfg = AnalysisConfig::from_env();
|
||||
assert!(!cfg.enabled);
|
||||
// OCR is the default engine (the Pi-friendly path).
|
||||
assert_eq!(cfg.backend, AnalysisBackend::Ocr);
|
||||
assert_eq!(cfg.ocr_detection_model, "/models/text-detection.rten");
|
||||
assert_eq!(cfg.ocr_recognition_model, "/models/text-recognition.rten");
|
||||
assert_eq!(cfg.workers, 1);
|
||||
assert_eq!(cfg.max_pixels, 1_000_000);
|
||||
assert_eq!(cfg.min_slice_height, 640);
|
||||
@@ -765,6 +964,51 @@ mod tests {
|
||||
std::env::remove_var("ANALYSIS_RESPONSE_FORMAT");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_backend_parses_and_defaults_to_ocr() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
for (raw, want) in [
|
||||
("ocr", AnalysisBackend::Ocr),
|
||||
("ocrs", AnalysisBackend::Ocr),
|
||||
("vision", AnalysisBackend::Vision),
|
||||
("llm", AnalysisBackend::Vision),
|
||||
("anything-else", AnalysisBackend::Ocr),
|
||||
] {
|
||||
std::env::set_var("ANALYSIS_BACKEND", raw);
|
||||
assert_eq!(AnalysisConfig::from_env().backend, want, "raw={raw}");
|
||||
}
|
||||
// Unset → OCR.
|
||||
std::env::remove_var("ANALYSIS_BACKEND");
|
||||
assert_eq!(AnalysisConfig::from_env().backend, AnalysisBackend::Ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn effective_backend_is_ocr_even_when_vision_requested() {
|
||||
// Vision is temporarily disabled: the worker always runs OCR no matter
|
||||
// what `ANALYSIS_BACKEND` parsed to. `backend` still reflects the raw
|
||||
// request (so the override is visible/loggable), but `effective_backend`
|
||||
// is the value the daemon actually dispatches through.
|
||||
let mut cfg = AnalysisConfig {
|
||||
backend: AnalysisBackend::Vision,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(cfg.effective_backend(), AnalysisBackend::Ocr);
|
||||
cfg.backend = AnalysisBackend::Ocr;
|
||||
assert_eq!(cfg.effective_backend(), AnalysisBackend::Ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ocr_model_paths_parse_from_env() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
std::env::set_var("OCRS_DETECTION_MODEL", "/opt/det.rten");
|
||||
std::env::set_var("OCRS_RECOGNITION_MODEL", "/opt/rec.rten");
|
||||
let cfg = AnalysisConfig::from_env();
|
||||
std::env::remove_var("OCRS_DETECTION_MODEL");
|
||||
std::env::remove_var("OCRS_RECOGNITION_MODEL");
|
||||
assert_eq!(cfg.ocr_detection_model, "/opt/det.rten");
|
||||
assert_eq!(cfg.ocr_recognition_model, "/opt/rec.rten");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_config_parses_from_env() {
|
||||
let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner());
|
||||
@@ -852,6 +1096,13 @@ mod tests {
|
||||
"BACKEND_URL",
|
||||
"BACKEND_PROXY_TIMEOUT_MS",
|
||||
"VISION_MANAGER_DATABASE_URL",
|
||||
// Compose-level deployment knobs (port publish interface + per-container
|
||||
// memory ceilings) — consumed by docker-compose.yml itself, never read
|
||||
// by the backend process, so they don't belong in its environment block.
|
||||
"FRONTEND_PUBLISH_ADDR",
|
||||
"BACKEND_MEM_LIMIT",
|
||||
"FRONTEND_MEM_LIMIT",
|
||||
"POSTGRES_MEM_LIMIT",
|
||||
];
|
||||
|
||||
/// Keys whose compose RHS is intentionally NOT a `${KEY...}`
|
||||
|
||||
@@ -234,12 +234,20 @@ impl BrowserManager {
|
||||
await_drain(&self.active, drain_deadline).await;
|
||||
|
||||
self.set_phase(RestartPhase::Restarting);
|
||||
let relaunch = {
|
||||
// Take the dead handle out under the lock, release the lock, THEN run
|
||||
// the (slow) Chromium teardown — so a worker that raced past the drain
|
||||
// can't block on `acquire()` behind one dead browser's close(). Mirrors
|
||||
// the idle reaper's take-drop-close ordering. Re-acquire to relaunch.
|
||||
let dead = {
|
||||
let mut guard = self.inner.lock().await;
|
||||
guard.shared = None;
|
||||
if let Some(handle) = guard.handle.take() {
|
||||
let _ = handle.close().await;
|
||||
}
|
||||
guard.handle.take()
|
||||
};
|
||||
if let Some(handle) = dead {
|
||||
let _ = handle.close().await;
|
||||
}
|
||||
let relaunch = {
|
||||
let mut guard = self.inner.lock().await;
|
||||
self.launch_into(&mut guard).await
|
||||
};
|
||||
|
||||
@@ -257,9 +265,14 @@ impl BrowserManager {
|
||||
/// Used on daemon shutdown. After this returns the next acquire will
|
||||
/// re-launch from scratch.
|
||||
pub async fn shutdown(&self) {
|
||||
let mut guard = self.inner.lock().await;
|
||||
guard.shared = None;
|
||||
if let Some(handle) = guard.handle.take() {
|
||||
// Take-then-drop-then-close: don't hold the lock across Chromium
|
||||
// teardown (see `invalidate` / the idle reaper).
|
||||
let handle = {
|
||||
let mut guard = self.inner.lock().await;
|
||||
guard.shared = None;
|
||||
guard.handle.take()
|
||||
};
|
||||
if let Some(handle) = handle {
|
||||
let _ = handle.close().await;
|
||||
}
|
||||
}
|
||||
@@ -278,9 +291,16 @@ impl BrowserManager {
|
||||
/// Idempotent: calling on an already-invalidated manager is a
|
||||
/// no-op.
|
||||
pub async fn invalidate(&self) {
|
||||
let mut guard = self.inner.lock().await;
|
||||
guard.shared = None;
|
||||
if let Some(handle) = guard.handle.take() {
|
||||
// Take the handle out under the lock, then release the lock BEFORE the
|
||||
// slow Chromium close() — otherwise every other worker's `acquire()`
|
||||
// serializes behind one dead browser's teardown. Matches the idle
|
||||
// reaper's ordering at the bottom of this file.
|
||||
let handle = {
|
||||
let mut guard = self.inner.lock().await;
|
||||
guard.shared = None;
|
||||
guard.handle.take()
|
||||
};
|
||||
if let Some(handle) = handle {
|
||||
let _ = handle.close().await;
|
||||
tracing::warn!("BrowserManager: handle invalidated — next acquire will relaunch");
|
||||
}
|
||||
|
||||
@@ -18,7 +18,7 @@ use uuid::Uuid;
|
||||
|
||||
use crate::crawler::detect::PageError;
|
||||
use crate::crawler::rate_limit::HostRateLimiters;
|
||||
use crate::crawler::safety::{fetch_stream, looks_like_image, DownloadAllowlist};
|
||||
use crate::crawler::safety::{ensure_public_target, fetch_stream, looks_like_image, DownloadAllowlist};
|
||||
use crate::crawler::session::{self, ChapterProbe};
|
||||
use crate::storage::{Storage, StorageError};
|
||||
|
||||
@@ -71,6 +71,13 @@ pub enum SyncOutcome {
|
||||
/// Session probe failed mid-sync (avatar selector missing on the
|
||||
/// chapter page). Caller should abort the whole crawler run.
|
||||
SessionExpired,
|
||||
/// The headless browser could not be acquired (down or mid-restart).
|
||||
/// Produced only by the queue dispatcher (which calls `acquire()`); it is
|
||||
/// an *infrastructure* outage, not a job failure, so the daemon returns the
|
||||
/// job to `pending` WITHOUT burning a retry attempt. `sync_chapter_content`
|
||||
/// itself never returns this — callers that already hold a lease can treat
|
||||
/// it as unreachable.
|
||||
BrowserUnavailable,
|
||||
}
|
||||
|
||||
/// Per-chapter max fetch attempts when TOR is configured. `N = 3` means
|
||||
@@ -95,6 +102,19 @@ enum ChapterFetchOutcome {
|
||||
PersistentTransient,
|
||||
}
|
||||
|
||||
/// Refuse to navigate Chromium at a chapter URL that points inside the
|
||||
/// deployment. The image-download path goes through `is_safe_url`, but
|
||||
/// `new_page` is a second network surface: a `chapter_sources` row whose
|
||||
/// host resolves to (or was crafted to be) a private IP would otherwise let
|
||||
/// the headless browser probe `postgres:5432`, the cloud metadata service,
|
||||
/// etc. Reuses the allowlist-free `ensure_public_target` (scheme +
|
||||
/// private-IP literal check) since there is no per-host allowlist for the
|
||||
/// scraped catalog itself.
|
||||
fn guard_nav_url(source_url: &str) -> anyhow::Result<()> {
|
||||
ensure_public_target(source_url)
|
||||
.map_err(|e| anyhow::anyhow!("refuse to navigate unsafe chapter URL {source_url}: {e}"))
|
||||
}
|
||||
|
||||
/// Single rate-limited Chromium navigation to the chapter URL,
|
||||
/// returning the page HTML. Extracted from `sync_chapter_content` so
|
||||
/// the recircuit loop can call it once per attempt.
|
||||
@@ -103,27 +123,36 @@ async fn fetch_chapter_html_once(
|
||||
rate: &HostRateLimiters,
|
||||
source_url: &str,
|
||||
) -> anyhow::Result<String> {
|
||||
guard_nav_url(source_url)?;
|
||||
rate.wait_for(source_url).await?;
|
||||
let page = browser
|
||||
.new_page(source_url)
|
||||
let page = crate::crawler::intercept::open_page(browser, source_url)
|
||||
.await
|
||||
.with_context(|| format!("open chapter page {source_url}"))?;
|
||||
crate::crawler::nav::wait_for_nav(&page)
|
||||
.await
|
||||
.context("wait for chapter nav")?;
|
||||
// Best-effort wait for the reader marker — same partial-render
|
||||
// race that bit the chapter-list parser can hit here. Timeout is
|
||||
// not an error; the chapter probe + parser sentinels still catch
|
||||
// real failures.
|
||||
let _ = crate::crawler::nav::wait_for_selector(
|
||||
&page,
|
||||
"a#pic_container",
|
||||
crate::crawler::nav::SELECTOR_TIMEOUT,
|
||||
// Close the tab on every exit path — a `?` on wait_for_nav / content()
|
||||
// would otherwise leak it (chromiumoxide doesn't close on drop).
|
||||
let closer = page.clone();
|
||||
crate::crawler::nav::close_after(
|
||||
async move {
|
||||
closer.close().await.ok();
|
||||
},
|
||||
async {
|
||||
crate::crawler::nav::wait_for_nav(&page)
|
||||
.await
|
||||
.context("wait for chapter nav")?;
|
||||
// Best-effort wait for the reader marker — same partial-render
|
||||
// race that bit the chapter-list parser can hit here. Timeout is
|
||||
// not an error; the chapter probe + parser sentinels still catch
|
||||
// real failures.
|
||||
let _ = crate::crawler::nav::wait_for_selector(
|
||||
&page,
|
||||
"a#pic_container",
|
||||
crate::crawler::nav::SELECTOR_TIMEOUT,
|
||||
)
|
||||
.await;
|
||||
page.content().await.context("read chapter html")
|
||||
},
|
||||
)
|
||||
.await;
|
||||
let html = page.content().await.context("read chapter html")?;
|
||||
page.close().await.ok();
|
||||
Ok(html)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Pure-over-IO loop: fetch + classify, up to `max_attempts` total
|
||||
@@ -214,6 +243,7 @@ pub async fn sync_chapter_content(
|
||||
force_refetch: bool,
|
||||
allowlist: &DownloadAllowlist,
|
||||
max_image_bytes: usize,
|
||||
max_images_per_chapter: usize,
|
||||
tor: Option<&crate::crawler::tor::TorController>,
|
||||
progress: Option<&crate::crawler::status::StatusHandle>,
|
||||
enqueue_analysis: bool,
|
||||
@@ -221,7 +251,8 @@ pub async fn sync_chapter_content(
|
||||
let started = std::time::Instant::now();
|
||||
let result = sync_chapter_content_inner(
|
||||
browser, db, storage, http, rate, chapter_id, manga_id, source_url,
|
||||
force_refetch, allowlist, max_image_bytes, tor, progress, enqueue_analysis,
|
||||
force_refetch, allowlist, max_image_bytes, max_images_per_chapter, tor, progress,
|
||||
enqueue_analysis,
|
||||
)
|
||||
.await;
|
||||
let duration_ms = started.elapsed().as_millis() as i64;
|
||||
@@ -271,6 +302,7 @@ async fn sync_chapter_content_inner(
|
||||
force_refetch: bool,
|
||||
allowlist: &DownloadAllowlist,
|
||||
max_image_bytes: usize,
|
||||
max_images_per_chapter: usize,
|
||||
tor: Option<&crate::crawler::tor::TorController>,
|
||||
// Optional live-status sink for the realtime page counter. The daemon
|
||||
// dispatcher passes the shared handle (the chapter has already been
|
||||
@@ -331,6 +363,16 @@ async fn sync_chapter_content_inner(
|
||||
if images.is_empty() {
|
||||
anyhow::bail!("no page images parsed from {source_url}");
|
||||
}
|
||||
// Bound total disk per chapter: the per-image byte cap doesn't stop a
|
||||
// hostile reader page from listing thousands of <img> tags. Ack failed
|
||||
// (the caller records it and backs off) rather than downloading them.
|
||||
if let Some(over) = image_count_over_cap(images.len(), max_images_per_chapter) {
|
||||
anyhow::bail!(
|
||||
"chapter at {source_url} lists {} page images, over the {} cap",
|
||||
over.count,
|
||||
over.cap
|
||||
);
|
||||
}
|
||||
|
||||
// Resolve image URLs against the chapter URL (they may be relative).
|
||||
let base = reqwest::Url::parse(source_url).context("parse chapter URL")?;
|
||||
@@ -407,6 +449,28 @@ pub(crate) struct StoredPage {
|
||||
/// "first 16 bytes" diagnostic in the error path is useful.
|
||||
const SNIFF_PREFIX_BYTES: usize = 64;
|
||||
|
||||
/// Bytes still admissible for the streaming tail after the sniff prefix
|
||||
/// has been drained. The prefix already counts against the per-image cap,
|
||||
/// so the tail budget is `max_image_bytes - prefix_len` using the
|
||||
/// **actual** drained length — never a constant — so `prefix_len + tail`
|
||||
/// can never exceed the cap.
|
||||
fn remaining_after_prefix(max_image_bytes: usize, prefix_len: usize) -> usize {
|
||||
max_image_bytes.saturating_sub(prefix_len)
|
||||
}
|
||||
|
||||
/// A rejected over-cap image count, carrying the numbers for the error.
|
||||
struct ImageCountOverCap {
|
||||
count: usize,
|
||||
cap: usize,
|
||||
}
|
||||
|
||||
/// `Some(..)` when `count` exceeds the per-chapter image cap. A `cap` of
|
||||
/// `0` disables the check (unbounded), matching the config's "0 = no cap"
|
||||
/// contract.
|
||||
fn image_count_over_cap(count: usize, cap: usize) -> Option<ImageCountOverCap> {
|
||||
(cap != 0 && count > cap).then_some(ImageCountOverCap { count, cap })
|
||||
}
|
||||
|
||||
/// Download a single page image, validate it's really an image, and
|
||||
/// stream it to storage. Returns the storage key + content type. Does
|
||||
/// not touch the DB — persistence is batched into one short transaction
|
||||
@@ -480,11 +544,18 @@ async fn download_and_store_page(
|
||||
// straight to storage. The cap is enforced via a running total in
|
||||
// the stream adapter so a server that omits Content-Length still
|
||||
// can't exhaust memory.
|
||||
// Budget the streaming tail against the bytes *actually* drained into
|
||||
// the prefix, not the constant `SNIFF_PREFIX_BYTES`. The prefix loop
|
||||
// appends whole chunks, so a single 16 KiB first chunk fills the
|
||||
// 64-byte sniff window in one drain — charging only 64 bytes would
|
||||
// then let the tail add another full `max_image_bytes`, storing up to
|
||||
// ~2× the cap. Capture the length before `prefix` is moved into the
|
||||
// stream below.
|
||||
let prefix_len = prefix.len();
|
||||
let prefix_stream = futures_util::stream::once(async move {
|
||||
Ok::<bytes::Bytes, StorageError>(prefix)
|
||||
});
|
||||
let prefix_len = SNIFF_PREFIX_BYTES.min(max_image_bytes);
|
||||
let mut remaining = max_image_bytes.saturating_sub(prefix_len);
|
||||
let mut remaining = remaining_after_prefix(max_image_bytes, prefix_len);
|
||||
let url_for_err = url.clone();
|
||||
let rest_stream = body.map(move |frame| match frame {
|
||||
Ok(chunk) => {
|
||||
@@ -524,15 +595,22 @@ pub(crate) async fn persist_pages(
|
||||
let mut tx = db.begin().await.context("open chapter sync tx")?;
|
||||
let mut page_ids: Vec<Uuid> = Vec::with_capacity(stored.len());
|
||||
let mut page_numbers: Vec<i32> = Vec::with_capacity(stored.len());
|
||||
// Pages whose row already existed and was overwritten by this re-crawl.
|
||||
// Their image changed but pages.id (and any analysis keyed to it) survived,
|
||||
// so the derived analysis is now stale and must be invalidated below.
|
||||
let mut updated_page_ids: Vec<Uuid> = Vec::new();
|
||||
for page in stored {
|
||||
let (id,): (Uuid,) = sqlx::query_as(
|
||||
// `xmax <> 0` on the RETURNING row distinguishes a conflict UPDATE
|
||||
// (existing row overwritten) from a fresh INSERT — the standard upsert
|
||||
// idiom. Cast through text/bigint since there is no direct xid<>int op.
|
||||
let (id, was_update): (Uuid, bool) = sqlx::query_as(
|
||||
"INSERT INTO pages (chapter_id, page_number, storage_key, content_type, size_bytes)
|
||||
VALUES ($1, $2, $3, $4, $5)
|
||||
ON CONFLICT (chapter_id, page_number) DO UPDATE
|
||||
SET storage_key = EXCLUDED.storage_key,
|
||||
content_type = EXCLUDED.content_type,
|
||||
size_bytes = EXCLUDED.size_bytes
|
||||
RETURNING id",
|
||||
RETURNING id, (xmax::text::bigint <> 0)",
|
||||
)
|
||||
.bind(chapter_id)
|
||||
.bind(page.page_number)
|
||||
@@ -544,6 +622,32 @@ pub(crate) async fn persist_pages(
|
||||
.with_context(|| format!("insert page row {}", page.page_number))?;
|
||||
page_ids.push(id);
|
||||
page_numbers.push(page.page_number);
|
||||
if was_update {
|
||||
updated_page_ids.push(id);
|
||||
}
|
||||
}
|
||||
// Invalidate stale analysis for re-crawled (overwritten) pages. Storage keys
|
||||
// are deterministic per page number, so the row survives the upsert and its
|
||||
// page_analysis / OCR / auto-tags / content-warnings would otherwise reflect
|
||||
// the OLD image — poisoning search results and the derived
|
||||
// manga_content_warnings. Clearing them (the DELETE on page_content_warnings
|
||||
// fires the mcw_* triggers so manga_content_warnings recomputes) drops the
|
||||
// pages back to "unanalyzed" so the enqueue below — and the admin
|
||||
// only-unanalyzed backfill — re-derive them. Newly inserted pages have no
|
||||
// prior analysis, so they're untouched.
|
||||
if !updated_page_ids.is_empty() {
|
||||
for table in [
|
||||
"page_ocr_text",
|
||||
"page_auto_tags",
|
||||
"page_content_warnings",
|
||||
"page_analysis",
|
||||
] {
|
||||
sqlx::query(&format!("DELETE FROM {table} WHERE page_id = ANY($1::uuid[])"))
|
||||
.bind(&updated_page_ids)
|
||||
.execute(&mut *tx)
|
||||
.await
|
||||
.with_context(|| format!("invalidate {table} for re-crawled pages"))?;
|
||||
}
|
||||
}
|
||||
// Drop any rows left over from a prior, larger crawl of this chapter
|
||||
// (re-crawl that now yields fewer pages). Without this the stale rows —
|
||||
@@ -609,6 +713,64 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::storage::LocalStorage;
|
||||
|
||||
#[test]
|
||||
fn guard_nav_url_rejects_private_and_loopback_targets() {
|
||||
// A chapter_sources row that resolves to / was crafted as an
|
||||
// internal target must be refused before Chromium navigates.
|
||||
for url in [
|
||||
"http://127.0.0.1:5432/",
|
||||
"http://169.254.169.254/latest/meta-data/",
|
||||
"http://10.0.0.1/chapter/1",
|
||||
"http://localhost:8080/",
|
||||
"file:///etc/passwd",
|
||||
] {
|
||||
assert!(guard_nav_url(url).is_err(), "must reject {url}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn guard_nav_url_allows_public_chapter_urls() {
|
||||
assert!(guard_nav_url("https://reader.example.com/chapter/42").is_ok());
|
||||
assert!(guard_nav_url("http://manga-host.test/c/1/p/2").is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tail_budget_uses_actual_prefix_length_not_constant() {
|
||||
// A single 16 KiB first chunk fills the 64-byte sniff window in one
|
||||
// drain, so the streaming tail budget must be `cap - 16KiB`, not
|
||||
// `cap - 64`. The constant-64 math previously admitted a further
|
||||
// ~full cap on top of the already-drained prefix (~2× overshoot).
|
||||
let cap = 20 * 1024;
|
||||
let prefix_len = 16 * 1024; // one real chunk, well over SNIFF_PREFIX_BYTES
|
||||
let remaining = remaining_after_prefix(cap, prefix_len);
|
||||
assert_eq!(remaining, cap - prefix_len);
|
||||
// Invariant: prefix + admitted tail never exceeds the cap.
|
||||
assert!(prefix_len + remaining <= cap);
|
||||
// And it's strictly tighter than the old constant-64 budget.
|
||||
assert!(remaining < cap - SNIFF_PREFIX_BYTES);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn image_count_cap_rejects_only_over_cap_and_respects_disable() {
|
||||
// Under / at the cap: accepted.
|
||||
assert!(image_count_over_cap(0, 2000).is_none());
|
||||
assert!(image_count_over_cap(2000, 2000).is_none());
|
||||
// Over the cap: rejected, carrying the numbers for the error.
|
||||
let over = image_count_over_cap(2001, 2000).expect("over cap");
|
||||
assert_eq!(over.count, 2001);
|
||||
assert_eq!(over.cap, 2000);
|
||||
// `0` disables the cap entirely (unbounded), matching config contract.
|
||||
assert!(image_count_over_cap(1_000_000, 0).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tail_budget_saturates_when_prefix_hits_cap() {
|
||||
// Body shorter than the sniff window: prefix drained == cap, tail
|
||||
// budget is zero (no negative underflow).
|
||||
assert_eq!(remaining_after_prefix(50, 50), 0);
|
||||
assert_eq!(remaining_after_prefix(50, 64), 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn cleanup_orphans_deletes_written_keys() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
@@ -1071,4 +1233,91 @@ mod tests {
|
||||
assert_eq!(fetch_n, 1);
|
||||
assert!(format!("{err:#}").contains("nav timeout"));
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn persist_pages_invalidates_stale_analysis_on_recrawl_update(pool: PgPool) {
|
||||
// Page storage keys are deterministic per page number, so a re-crawl
|
||||
// overwrites the image but keeps pages.id. Analysis keyed to page_id
|
||||
// would otherwise stay stale (old OCR/search_doc/warnings). persist_pages
|
||||
// must clear the derived analysis for updated pages so it re-derives.
|
||||
let manga_id = Uuid::new_v4();
|
||||
let chapter_id = Uuid::new_v4();
|
||||
sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, 'T')")
|
||||
.bind(manga_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
sqlx::query("INSERT INTO chapters (id, manga_id, number) VALUES ($1, $2, 1)")
|
||||
.bind(chapter_id)
|
||||
.bind(manga_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let v1 = vec![StoredPage {
|
||||
page_number: 1,
|
||||
storage_key: "k/0001.jpg".into(),
|
||||
content_type: "image/jpeg".into(),
|
||||
size_bytes: 10,
|
||||
}];
|
||||
persist_pages(&pool, chapter_id, &v1, false).await.unwrap();
|
||||
let page_id: Uuid = sqlx::query_scalar(
|
||||
"SELECT id FROM pages WHERE chapter_id = $1 AND page_number = 1",
|
||||
)
|
||||
.bind(chapter_id)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Simulate a completed prior analysis pass for that page.
|
||||
sqlx::query(
|
||||
"INSERT INTO page_analysis (page_id, status, is_nsfw, analyzed_at) \
|
||||
VALUES ($1, 'done', true, now())",
|
||||
)
|
||||
.bind(page_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
sqlx::query("INSERT INTO page_ocr_text (page_id, kind, text) VALUES ($1, 'speech', 'stale')")
|
||||
.bind(page_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
sqlx::query("INSERT INTO page_content_warnings (page_id, warning) VALUES ($1, 'gore')")
|
||||
.bind(page_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Re-crawl page 1 with a new image (same deterministic key, new size).
|
||||
let v2 = vec![StoredPage {
|
||||
page_number: 1,
|
||||
storage_key: "k/0001.jpg".into(),
|
||||
content_type: "image/jpeg".into(),
|
||||
size_bytes: 999,
|
||||
}];
|
||||
persist_pages(&pool, chapter_id, &v2, false).await.unwrap();
|
||||
|
||||
// The page row survived (deterministic key).
|
||||
let same_id: Uuid = sqlx::query_scalar(
|
||||
"SELECT id FROM pages WHERE chapter_id = $1 AND page_number = 1",
|
||||
)
|
||||
.bind(chapter_id)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(same_id, page_id, "re-crawl keeps the page id");
|
||||
|
||||
// ...but its stale analysis is cleared so it gets re-derived.
|
||||
for table in ["page_analysis", "page_ocr_text", "page_content_warnings"] {
|
||||
let n: i64 = sqlx::query_scalar(&format!(
|
||||
"SELECT COUNT(*) FROM {table} WHERE page_id = $1"
|
||||
))
|
||||
.bind(page_id)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(n, 0, "{table} must be invalidated when the re-crawl updates the page");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -66,6 +66,43 @@ const LEASE_DURATION: Duration = Duration::from_secs(60);
|
||||
/// the lease window leaves two missed-beat's slack before expiry.
|
||||
const LEASE_HEARTBEAT: Duration = Duration::from_secs(20);
|
||||
|
||||
/// Consecutive failed lease renews the heartbeat tolerates before it gives up
|
||||
/// and signals the worker to abandon the in-flight dispatch. At
|
||||
/// [`LEASE_HEARTBEAT`] spacing this is ~1 lease window of DB flakiness — past
|
||||
/// that the lease has very likely lapsed and another worker may re-lease the
|
||||
/// job, so continuing to crawl it is wasted (and duplicated) work.
|
||||
const MAX_HEARTBEAT_RENEW_FAILURES: u32 = 3;
|
||||
|
||||
/// Whether the heartbeat should abandon the job after `consecutive_failures`
|
||||
/// failed renews. Split out so the escalation threshold is unit-testable.
|
||||
fn should_abort_after_renew_failures(consecutive_failures: u32) -> bool {
|
||||
consecutive_failures >= MAX_HEARTBEAT_RENEW_FAILURES
|
||||
}
|
||||
|
||||
/// How long a worker waits after a `BrowserUnavailable` outcome before looping
|
||||
/// back to lease again. The job was released (not failed) so it stays pending;
|
||||
/// this backoff keeps the worker from hot-looping lease→acquire→release while
|
||||
/// the browser is down or mid-restart.
|
||||
const BROWSER_UNAVAILABLE_BACKOFF: Duration = Duration::from_secs(5);
|
||||
|
||||
/// Longest an idle worker waits between lease polls. Bounds the exponential
|
||||
/// [`idle_backoff`] so a worker still notices freshly-enqueued work reasonably
|
||||
/// soon after a quiet spell.
|
||||
const IDLE_BACKOFF_CAP: Duration = Duration::from_secs(30);
|
||||
|
||||
/// Backoff for a worker that keeps finding no work: 1s, 2s, 4s, … capped at
|
||||
/// [`IDLE_BACKOFF_CAP`]. `consecutive_empty` is the number of empty polls seen
|
||||
/// so far (0 on the first miss); reset to 0 the moment a job is leased. Replaces
|
||||
/// the old flat 1s sleep so an idle daemon isn't firing a row-locking `SELECT …
|
||||
/// FOR UPDATE SKIP LOCKED` lease query every second per worker.
|
||||
fn idle_backoff(consecutive_empty: u32) -> Duration {
|
||||
let cap = IDLE_BACKOFF_CAP.as_secs();
|
||||
// 1 << n grows the interval; saturate to the cap once the shift overflows
|
||||
// or the value exceeds the cap.
|
||||
let secs = 1u64.checked_shl(consecutive_empty).unwrap_or(cap).min(cap);
|
||||
Duration::from_secs(secs)
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
pub trait MetadataPass: Send + Sync {
|
||||
async fn run(&self) -> anyhow::Result<pipeline::MetadataStats>;
|
||||
@@ -301,9 +338,9 @@ impl CronContext {
|
||||
}
|
||||
Err(e) => tracing::error!(?e, "cron: enqueue_bookmarked_pending failed"),
|
||||
}
|
||||
match jobs::reap_done(pool, retention_days).await {
|
||||
Ok(n) => tracing::info!(reaped = n, "cron: done-job reaper finished"),
|
||||
Err(e) => tracing::error!(?e, "cron: done-job reaper failed"),
|
||||
match jobs::reap_terminal(pool, retention_days).await {
|
||||
Ok(n) => tracing::info!(reaped = n, "cron: terminal-job reaper finished"),
|
||||
Err(e) => tracing::error!(?e, "cron: terminal-job reaper failed"),
|
||||
}
|
||||
match crate::repo::crawl_metrics::reap(pool, metrics_retention_days).await {
|
||||
Ok(n) => tracing::info!(reaped = n, "cron: crawl-metrics reaper finished"),
|
||||
@@ -356,6 +393,9 @@ struct WorkerContext {
|
||||
|
||||
impl WorkerContext {
|
||||
async fn run(self) {
|
||||
// Consecutive empty lease polls, driving the idle backoff. Reset to 0
|
||||
// the moment any job is leased.
|
||||
let mut idle_streak: u32 = 0;
|
||||
loop {
|
||||
if self.cancel.is_cancelled() {
|
||||
tracing::info!(worker = self.id, "worker: shutdown");
|
||||
@@ -385,11 +425,14 @@ impl WorkerContext {
|
||||
}
|
||||
};
|
||||
let Some(lease) = leases.into_iter().next() else {
|
||||
let backoff = idle_backoff(idle_streak);
|
||||
idle_streak = idle_streak.saturating_add(1);
|
||||
tokio::select! {
|
||||
_ = tokio::time::sleep(Duration::from_secs(1)) => continue,
|
||||
_ = tokio::time::sleep(backoff) => continue,
|
||||
_ = self.cancel.cancelled() => return,
|
||||
}
|
||||
};
|
||||
idle_streak = 0;
|
||||
self.process_lease(lease).await;
|
||||
}
|
||||
}
|
||||
@@ -413,18 +456,34 @@ impl WorkerContext {
|
||||
// dispatch runs, so a slow-but-healthy job is never re-leased and
|
||||
// never inflates `attempts` toward `max_attempts`. Stops itself
|
||||
// once the job is no longer ours (renew returns false).
|
||||
// Signalled by the heartbeat if it gives up after too many consecutive
|
||||
// renew failures, so the worker can abandon a dispatch whose lease has
|
||||
// very likely lapsed (rather than crawl a job another worker may now own).
|
||||
let hb_lost = CancellationToken::new();
|
||||
let heartbeat = {
|
||||
let hb_pool = self.pool.clone();
|
||||
let hb_id = lease.id;
|
||||
let hb_gen = lease.lease_generation;
|
||||
let hb_lost = hb_lost.clone();
|
||||
tokio::spawn(async move {
|
||||
let mut failures: u32 = 0;
|
||||
loop {
|
||||
tokio::time::sleep(LEASE_HEARTBEAT).await;
|
||||
match jobs::renew(&hb_pool, hb_id, hb_gen, LEASE_DURATION).await {
|
||||
Ok(true) => {}
|
||||
Ok(true) => failures = 0,
|
||||
Ok(false) => break,
|
||||
Err(e) => {
|
||||
tracing::warn!(lease_id = %hb_id, ?e, "heartbeat renew failed");
|
||||
failures += 1;
|
||||
tracing::warn!(lease_id = %hb_id, failures, ?e, "heartbeat renew failed");
|
||||
if should_abort_after_renew_failures(failures) {
|
||||
tracing::error!(
|
||||
lease_id = %hb_id,
|
||||
failures,
|
||||
"heartbeat lost the lease after repeated renew failures — signalling abandon"
|
||||
);
|
||||
hb_lost.cancel();
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -462,6 +521,20 @@ impl WorkerContext {
|
||||
);
|
||||
return;
|
||||
}
|
||||
_ = hb_lost.cancelled() => {
|
||||
// The heartbeat gave up renewing: the lease has very likely
|
||||
// expired and may already be re-leased elsewhere. Abandon the
|
||||
// dispatch rather than keep crawling a job we no longer own. Do
|
||||
// NOT ack/release — a generation-guarded write would no-op, and
|
||||
// another worker may now hold this lease.
|
||||
heartbeat.abort();
|
||||
tracing::error!(
|
||||
worker = self.id,
|
||||
lease_id = %lease.id,
|
||||
"worker: abandoning dispatch — lease lost (heartbeat renew failures)"
|
||||
);
|
||||
return;
|
||||
}
|
||||
o = tokio::time::timeout(self.job_timeout, dispatch) => o,
|
||||
};
|
||||
heartbeat.abort();
|
||||
@@ -502,6 +575,24 @@ impl WorkerContext {
|
||||
self.status.poke();
|
||||
let _ = jobs::release(&self.pool, lease.id, lease.lease_generation).await;
|
||||
}
|
||||
Ok(Ok(SyncOutcome::BrowserUnavailable)) => {
|
||||
// Infrastructure outage, not a job failure: the browser was
|
||||
// down or mid-restart when the dispatcher tried to acquire it.
|
||||
// Return the job to `pending` WITHOUT burning an attempt (like
|
||||
// the cancel/session paths) so an outage doesn't chew the whole
|
||||
// backlog to `dead`, then back off to avoid hot-looping while
|
||||
// the browser recovers.
|
||||
tracing::warn!(
|
||||
worker = self.id,
|
||||
lease_id = %lease.id,
|
||||
"worker: browser unavailable — released lease without burning an attempt"
|
||||
);
|
||||
let _ = jobs::release(&self.pool, lease.id, lease.lease_generation).await;
|
||||
tokio::select! {
|
||||
_ = tokio::time::sleep(BROWSER_UNAVAILABLE_BACKOFF) => {}
|
||||
_ = self.cancel.cancelled() => {}
|
||||
}
|
||||
}
|
||||
Ok(Err(e)) => {
|
||||
tracing::warn!(
|
||||
worker = self.id,
|
||||
@@ -750,6 +841,37 @@ mod tests {
|
||||
Utc.with_ymd_and_hms(y, mo, d, h, mi, 0).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn heartbeat_aborts_only_after_threshold_consecutive_failures() {
|
||||
// A blip or two is tolerated; sustained failures escalate to abandon.
|
||||
assert!(!should_abort_after_renew_failures(0));
|
||||
assert!(!should_abort_after_renew_failures(1));
|
||||
assert!(!should_abort_after_renew_failures(MAX_HEARTBEAT_RENEW_FAILURES - 1));
|
||||
assert!(should_abort_after_renew_failures(MAX_HEARTBEAT_RENEW_FAILURES));
|
||||
assert!(should_abort_after_renew_failures(MAX_HEARTBEAT_RENEW_FAILURES + 1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn idle_backoff_grows_then_caps() {
|
||||
// First miss is a short 1s poll; the interval doubles each empty poll…
|
||||
assert_eq!(idle_backoff(0), Duration::from_secs(1));
|
||||
assert_eq!(idle_backoff(1), Duration::from_secs(2));
|
||||
assert_eq!(idle_backoff(2), Duration::from_secs(4));
|
||||
assert_eq!(idle_backoff(3), Duration::from_secs(8));
|
||||
assert_eq!(idle_backoff(4), Duration::from_secs(16));
|
||||
// …and saturates at the cap rather than growing unbounded.
|
||||
assert_eq!(idle_backoff(5), IDLE_BACKOFF_CAP);
|
||||
assert_eq!(idle_backoff(100), IDLE_BACKOFF_CAP);
|
||||
// Never exceeds the cap and is monotonic non-decreasing.
|
||||
let mut prev = Duration::ZERO;
|
||||
for n in 0..40 {
|
||||
let b = idle_backoff(n);
|
||||
assert!(b >= prev, "backoff must be non-decreasing at n={n}");
|
||||
assert!(b <= IDLE_BACKOFF_CAP, "backoff must never exceed the cap");
|
||||
prev = b;
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn next_fire_in_utc_at_midnight_advances_one_day() {
|
||||
let now = dt_utc(2026, 5, 25, 12, 0); // noon UTC
|
||||
|
||||
301
backend/src/crawler/intercept.rs
Normal file
301
backend/src/crawler/intercept.rs
Normal file
@@ -0,0 +1,301 @@
|
||||
//! Optional CDP-level SSRF guard for headless-browser navigations.
|
||||
//!
|
||||
//! The reqwest clients get a DNS-filtering resolver (see
|
||||
//! [`crate::crawler::safety::SafeResolver`]) so an image/API host that
|
||||
//! resolves to an internal IP is refused at connect time. Chromium, however,
|
||||
//! does its **own** DNS and connection handling, so that resolver can't see
|
||||
//! browser navigations. Without a second guard, a scraped chapter page that
|
||||
//! `302`-redirects the browser to `http://127.0.0.1:5432/` (or an
|
||||
//! attacker-owned hostname that resolves to `169.254.169.254`) would be
|
||||
//! loaded and parsed as if it were catalog content.
|
||||
//!
|
||||
//! This module installs CDP `Fetch` interception on a page so **every** request
|
||||
//! it issues — the main-frame Document, its redirects, *and every subresource*
|
||||
//! (`<img>`, `fetch()`/XHR, media, …) — is re-validated through the same
|
||||
//! [`ensure_public_target`] + resolved-IP check the reqwest paths use, failing
|
||||
//! any that target an internal address. Intercepting only the Document would
|
||||
//! leave a scraped page free to pull `<img src="http://169.254.169.254/…">` or
|
||||
//! `fetch('http://postgres:5432')` straight past the guard, so no resource type
|
||||
//! is exempt. (WebSocket handshakes are not surfaced by CDP `Fetch`, so `ws://`
|
||||
//! internal targets remain out of this hook's reach — the reqwest-layer
|
||||
//! resolver does not see them either; documented as a known gap.)
|
||||
//!
|
||||
//! **Opt-in / default-off.** Enabling `Fetch` means every intercepted request
|
||||
//! *must* be resolved by a live handler or the navigation hangs, so this is a
|
||||
//! fragile hook in the crawler's critical path. It ships behind
|
||||
//! `CRAWLER_SSRF_INTERCEPT` (default `false`) and the wiring has **not** been
|
||||
//! exercised against a real Chromium in CI — validate with a manual crawl
|
||||
//! before enabling in production. When disabled, [`open_page`] is byte-for-byte
|
||||
//! the previous `browser.new_page(url)` behavior. When enabled, the guard is
|
||||
//! **fail-closed**: if interception can't be installed, [`open_page`] closes the
|
||||
//! blank page and returns the error rather than navigating unguarded.
|
||||
|
||||
use std::net::IpAddr;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
|
||||
use chromiumoxide::browser::Browser;
|
||||
use chromiumoxide::cdp::browser_protocol::fetch::{
|
||||
ContinueRequestParams, EnableParams, EventRequestPaused, FailRequestParams, RequestPattern,
|
||||
RequestStage,
|
||||
};
|
||||
use chromiumoxide::cdp::browser_protocol::network::ErrorReason;
|
||||
use chromiumoxide::error::Result as CdpResult;
|
||||
use chromiumoxide::Page;
|
||||
use futures_util::StreamExt;
|
||||
use reqwest::Url;
|
||||
|
||||
use crate::crawler::safety::{ensure_public_target, is_private_ip};
|
||||
|
||||
/// Process-wide toggle, set once at startup from `CRAWLER_SSRF_INTERCEPT`.
|
||||
/// A single boot-time flag (rather than threading a param through every
|
||||
/// crawler navigation signature) keeps the off-path a no-op.
|
||||
static ENABLED: AtomicBool = AtomicBool::new(false);
|
||||
|
||||
pub fn set_enabled(on: bool) {
|
||||
ENABLED.store(on, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
pub fn is_enabled() -> bool {
|
||||
ENABLED.load(Ordering::Relaxed)
|
||||
}
|
||||
|
||||
/// What to do with a navigation URL, decided without DNS where possible so the
|
||||
/// security-critical branching is unit-testable.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub(crate) enum Verdict {
|
||||
/// Safe to continue (non-network scheme, or a public IP literal).
|
||||
Allow,
|
||||
/// Refuse — bad scheme handled elsewhere, localhost, or a private IP literal.
|
||||
Block,
|
||||
/// A hostname that must be resolved to decide (DNS-rebinding check).
|
||||
ResolveHost { host: String, port: u16 },
|
||||
}
|
||||
|
||||
/// Pure decision over a URL string. `data:` / `blob:` / `about:` and other
|
||||
/// non-http(s) schemes are allowed (not a network-SSRF vector); http(s) with a
|
||||
/// private/loopback/localhost literal is blocked; an http(s) hostname needs
|
||||
/// resolution.
|
||||
pub(crate) fn verdict(url: &str) -> Verdict {
|
||||
let Ok(parsed) = Url::parse(url) else {
|
||||
// Unparseable — let Chromium reject it; not our call to make.
|
||||
return Verdict::Allow;
|
||||
};
|
||||
match parsed.scheme() {
|
||||
"http" | "https" => {}
|
||||
// Non-http(s) subresource schemes: block the ones that can reach the
|
||||
// local filesystem or pivot to another protocol; continue the rest
|
||||
// (data:/blob:/about: and Chrome-internal schemes) which legitimate page
|
||||
// loads depend on — blocking those would wedge navigation.
|
||||
"file" | "ftp" | "gopher" => return Verdict::Block,
|
||||
_ => return Verdict::Allow,
|
||||
}
|
||||
// Literal private IP / localhost / missing host → block (string check).
|
||||
if ensure_public_target(url).is_err() {
|
||||
return Verdict::Block;
|
||||
}
|
||||
match parsed.host_str() {
|
||||
Some(host) if !is_ip_literal(host) => Verdict::ResolveHost {
|
||||
host: host.to_string(),
|
||||
port: parsed.port_or_known_default().unwrap_or(80),
|
||||
},
|
||||
// Public IP literal (ensure_public_target already passed it).
|
||||
_ => Verdict::Allow,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `host` (as reqwest's `host_str()` yields it) is an IP literal.
|
||||
/// IPv6 literals arrive bracketed (`[::1]`), which don't parse as `IpAddr`
|
||||
/// directly — strip the brackets first.
|
||||
fn is_ip_literal(host: &str) -> bool {
|
||||
let unbracketed = host
|
||||
.strip_prefix('[')
|
||||
.and_then(|s| s.strip_suffix(']'))
|
||||
.unwrap_or(host);
|
||||
unbracketed.parse::<IpAddr>().is_ok()
|
||||
}
|
||||
|
||||
/// Async decision: resolves hostnames and blocks if any resolved address is
|
||||
/// private (DNS rebinding). A resolution failure is *not* treated as blocked —
|
||||
/// Chromium won't be able to connect either, so there's nothing to exfiltrate.
|
||||
pub(crate) async fn is_blocked(url: &str) -> bool {
|
||||
match verdict(url) {
|
||||
Verdict::Allow => false,
|
||||
Verdict::Block => true,
|
||||
Verdict::ResolveHost { host, port } => match tokio::net::lookup_host((host.as_str(), port))
|
||||
.await
|
||||
{
|
||||
Ok(addrs) => addrs.map(|a| a.ip()).any(|ip| is_private_ip(&ip)),
|
||||
Err(_) => false,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Open a page for navigation. With interception off, this is exactly
|
||||
/// `browser.new_page(url)`. With it on, the page is created blank (no network),
|
||||
/// the navigation guard is installed, and only then does it navigate — so the
|
||||
/// initial request and any redirects pass through the guard.
|
||||
pub async fn open_page(browser: &Browser, url: &str) -> CdpResult<Page> {
|
||||
if !is_enabled() {
|
||||
return browser.new_page(url).await;
|
||||
}
|
||||
let page = browser.new_page("about:blank").await?;
|
||||
if let Err(e) = install_navigation_guard(&page).await {
|
||||
// Fail closed: an unguarded page could be redirected (or pull a
|
||||
// subresource) to an internal target, so refuse to navigate. Close the
|
||||
// blank page and surface the error so the caller aborts this fetch.
|
||||
tracing::warn!(url = %url, error = %e, "SSRF navigation guard failed to install; aborting navigation (fail-closed)");
|
||||
let _ = page.close().await;
|
||||
return Err(e);
|
||||
}
|
||||
page.goto(url).await?;
|
||||
Ok(page)
|
||||
}
|
||||
|
||||
/// The `Fetch` interception patterns to register. A single pattern with **no**
|
||||
/// `resource_type` constraint matches every request the page makes — Document,
|
||||
/// Image, Fetch/XHR, media, everything — so subresources to internal targets are
|
||||
/// re-validated too, not only the main-frame navigation. Restricting this to
|
||||
/// `ResourceType::Document` (the previous behavior) left every `<img>`/`fetch()`
|
||||
/// subresource unguarded, which is the SSRF hole this closes. Pinned to the
|
||||
/// request stage so each request is paused once, before it leaves the browser.
|
||||
fn interception_patterns() -> Vec<RequestPattern> {
|
||||
vec![RequestPattern::builder()
|
||||
.request_stage(RequestStage::Request)
|
||||
.build()]
|
||||
}
|
||||
|
||||
/// Enable `Fetch` for all requests on `page` and spawn a task that
|
||||
/// continues/fails each paused request per [`is_blocked`].
|
||||
async fn install_navigation_guard(page: &Page) -> CdpResult<()> {
|
||||
page.execute(EnableParams {
|
||||
patterns: Some(interception_patterns()),
|
||||
handle_auth_requests: None,
|
||||
})
|
||||
.await?;
|
||||
|
||||
let mut paused = page.event_listener::<EventRequestPaused>().await?;
|
||||
let handler_page = page.clone();
|
||||
tokio::spawn(async move {
|
||||
while let Some(ev) = paused.next().await {
|
||||
let request_id = ev.request_id.clone();
|
||||
let outcome = if is_blocked(&ev.request.url).await {
|
||||
tracing::warn!(
|
||||
url = %ev.request.url,
|
||||
"SSRF guard blocked browser navigation to an internal target"
|
||||
);
|
||||
handler_page
|
||||
.execute(FailRequestParams::new(request_id, ErrorReason::BlockedByClient))
|
||||
.await
|
||||
.map(|_| ())
|
||||
} else {
|
||||
handler_page
|
||||
.execute(ContinueRequestParams::new(request_id))
|
||||
.await
|
||||
.map(|_| ())
|
||||
};
|
||||
if let Err(e) = outcome {
|
||||
// The page/session is gone (page closed) — the stream will end
|
||||
// too; stop handling so the task exits rather than spins.
|
||||
tracing::debug!(error = %e, "fetch interceptor: continue/fail failed, ending handler");
|
||||
break;
|
||||
}
|
||||
}
|
||||
});
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn verdict_allows_non_network_schemes() {
|
||||
for url in ["about:blank", "data:text/html,hi", "blob:abc", "chrome://version"] {
|
||||
assert_eq!(verdict(url), Verdict::Allow, "{url} should be allowed");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn verdict_blocks_dangerous_non_http_schemes() {
|
||||
// file:/ftp:/gopher: can reach the local filesystem or pivot protocols;
|
||||
// block them explicitly rather than falling through to Allow.
|
||||
for url in ["file:///etc/passwd", "ftp://internal/x", "gopher://internal:70/1"] {
|
||||
assert_eq!(verdict(url), Verdict::Block, "{url} should be blocked");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn verdict_blocks_private_ip_literals_and_localhost() {
|
||||
for url in [
|
||||
"http://127.0.0.1/",
|
||||
"http://10.0.0.1/",
|
||||
"http://192.168.1.1:5432/",
|
||||
"http://169.254.169.254/latest/meta-data/",
|
||||
"http://[::1]/",
|
||||
"http://[::ffff:127.0.0.1]/",
|
||||
"http://localhost/",
|
||||
"https://[fd00::1]/",
|
||||
] {
|
||||
assert_eq!(verdict(url), Verdict::Block, "{url} should be blocked");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn verdict_allows_public_ip_literals() {
|
||||
for url in ["http://93.184.216.34/", "https://[2606:4700:4700::1111]/"] {
|
||||
assert_eq!(verdict(url), Verdict::Allow, "{url} should be allowed");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn verdict_defers_hostnames_to_resolution() {
|
||||
assert_eq!(
|
||||
verdict("https://cdn.example.com/img.jpg"),
|
||||
Verdict::ResolveHost {
|
||||
host: "cdn.example.com".to_string(),
|
||||
port: 443
|
||||
}
|
||||
);
|
||||
assert_eq!(
|
||||
verdict("http://catalog.test:8080/list"),
|
||||
Verdict::ResolveHost {
|
||||
host: "catalog.test".to_string(),
|
||||
port: 8080
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn is_blocked_true_for_private_literal_no_dns() {
|
||||
assert!(is_blocked("http://169.254.169.254/").await);
|
||||
assert!(!is_blocked("http://93.184.216.34/").await);
|
||||
// Non-network scheme is always allowed through.
|
||||
assert!(!is_blocked("about:blank").await);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn interception_covers_all_resource_types() {
|
||||
// Regression guard for the SSRF subresource hole: the CDP Fetch pattern
|
||||
// must NOT be constrained to Document, or `<img>`/`fetch()`/XHR
|
||||
// subresources to internal targets slip past `is_blocked`. A pattern
|
||||
// with no `resource_type` set intercepts every request type.
|
||||
let patterns = interception_patterns();
|
||||
assert_eq!(patterns.len(), 1);
|
||||
assert!(
|
||||
patterns[0].resource_type.is_none(),
|
||||
"interception must cover all resource types (subresources included), not just Document"
|
||||
);
|
||||
assert_eq!(patterns[0].request_stage, Some(RequestStage::Request));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn enabled_toggle_roundtrips() {
|
||||
// Global; restore afterwards so other tests see the default.
|
||||
let prev = is_enabled();
|
||||
set_enabled(true);
|
||||
assert!(is_enabled());
|
||||
set_enabled(false);
|
||||
assert!(!is_enabled());
|
||||
set_enabled(prev);
|
||||
}
|
||||
}
|
||||
@@ -132,14 +132,17 @@ fn backoff_for(attempts: i32) -> Duration {
|
||||
/// `Skipped`. The slot frees again once the previous job leaves the
|
||||
/// in-flight states (done/failed/dead), so a re-enqueue after a force
|
||||
/// refetch succeeds.
|
||||
pub async fn enqueue(pool: &PgPool, payload: &JobPayload) -> sqlx::Result<EnqueueResult> {
|
||||
pub async fn enqueue<'e, E>(executor: E, payload: &JobPayload) -> sqlx::Result<EnqueueResult>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
let json = serde_json::to_value(payload).expect("JobPayload is always serializable");
|
||||
let id: Option<Uuid> = sqlx::query_scalar(
|
||||
"INSERT INTO crawler_jobs (payload) VALUES ($1) \
|
||||
ON CONFLICT DO NOTHING RETURNING id",
|
||||
)
|
||||
.bind(json)
|
||||
.fetch_optional(pool)
|
||||
.fetch_optional(executor)
|
||||
.await?;
|
||||
Ok(match id {
|
||||
Some(id) => EnqueueResult::Inserted(id),
|
||||
@@ -422,7 +425,10 @@ pub async fn release(
|
||||
/// the original worker becomes a no-op. Returns the attempt to
|
||||
/// `pending` and refunds the retry attempt (the operator click
|
||||
/// isn't a job-level failure).
|
||||
pub async fn release_unowned(pool: &PgPool, lease_id: Uuid) -> sqlx::Result<()> {
|
||||
pub async fn release_unowned<'e, E>(executor: E, lease_id: Uuid) -> sqlx::Result<()>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
let res = sqlx::query(
|
||||
"UPDATE crawler_jobs \
|
||||
SET state = 'pending', leased_until = NULL, \
|
||||
@@ -432,7 +438,7 @@ pub async fn release_unowned(pool: &PgPool, lease_id: Uuid) -> sqlx::Result<()>
|
||||
WHERE id = $1 AND state = 'running'",
|
||||
)
|
||||
.bind(lease_id)
|
||||
.execute(pool)
|
||||
.execute(executor)
|
||||
.await?;
|
||||
if res.rows_affected() == 0 {
|
||||
tracing::warn!(
|
||||
@@ -479,22 +485,43 @@ pub async fn reclaim_orphaned(pool: &PgPool) -> sqlx::Result<u64> {
|
||||
Ok(result.rows_affected())
|
||||
}
|
||||
|
||||
/// Delete `done` jobs whose `updated_at` is older than `retention_days`
|
||||
/// days. `0` disables the reaper without touching the table. Returns the
|
||||
/// number of rows removed.
|
||||
pub async fn reap_done(pool: &PgPool, retention_days: u32) -> sqlx::Result<u64> {
|
||||
/// Delete **terminal** jobs (`done` or `dead`) whose `updated_at` is older
|
||||
/// than `retention_days` days. Both states are end-of-life — `done`
|
||||
/// succeeded, `dead` exhausted its retries — and neither is ever leased
|
||||
/// again, so reaping only `done` left `dead` rows to accumulate forever.
|
||||
/// `pending` / `running` are active and never touched. `0` disables the
|
||||
/// reaper without touching the table. Returns the number of rows removed.
|
||||
pub async fn reap_terminal(pool: &PgPool, retention_days: u32) -> sqlx::Result<u64> {
|
||||
if retention_days == 0 {
|
||||
return Ok(0);
|
||||
}
|
||||
let result = sqlx::query(
|
||||
"DELETE FROM crawler_jobs \
|
||||
WHERE state = 'done' \
|
||||
AND updated_at < now() - ($1::bigint || ' days')::interval",
|
||||
)
|
||||
.bind(retention_days as i64)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(result.rows_affected())
|
||||
// Delete in bounded batches rather than one statement. A single unbatched
|
||||
// DELETE over an unbounded backlog takes a long-held lock and one huge
|
||||
// transaction; batching keeps each delete short (index-backed by
|
||||
// crawler_jobs_terminal_reap_idx, 0040) so the reaper never pins the table
|
||||
// (audit M5). Loop until a batch comes back short.
|
||||
const BATCH: i64 = 5_000;
|
||||
let mut total: u64 = 0;
|
||||
loop {
|
||||
let result = sqlx::query(
|
||||
"DELETE FROM crawler_jobs \
|
||||
WHERE ctid IN ( \
|
||||
SELECT ctid FROM crawler_jobs \
|
||||
WHERE state IN ('done', 'dead') \
|
||||
AND updated_at < now() - ($1::bigint || ' days')::interval \
|
||||
LIMIT $2 )",
|
||||
)
|
||||
.bind(retention_days as i64)
|
||||
.bind(BATCH)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
let n = result.rows_affected();
|
||||
total += n;
|
||||
if n < BATCH as u64 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Ok(total)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -19,6 +19,7 @@ pub mod content;
|
||||
pub mod daemon;
|
||||
pub mod detect;
|
||||
pub mod diff;
|
||||
pub mod intercept;
|
||||
pub mod jobs;
|
||||
pub mod nav;
|
||||
pub mod pipeline;
|
||||
|
||||
@@ -85,6 +85,28 @@ pub async fn wait_for_selector(
|
||||
/// under a second.
|
||||
pub const SELECTOR_TIMEOUT: Duration = Duration::from_secs(10);
|
||||
|
||||
/// Run `body` to completion, then run `close` — on *every* exit path,
|
||||
/// including an early `?` / `return` inside `body`. chromiumoxide's `Page`
|
||||
/// does **not** close its CDP target on drop, so a fetch helper that opens a
|
||||
/// page with `browser.new_page(...)` and then bails via `?` on a nav /
|
||||
/// content-read error would leak a browser tab for the process's lifetime.
|
||||
/// Wrapping the fallible work in `close_after` — with `close` built from a
|
||||
/// clone of the page — guarantees the tab is closed regardless of how `body`
|
||||
/// returns.
|
||||
///
|
||||
/// Both arguments are pre-built futures, so this stays generic over the
|
||||
/// body's return type (the anyhow and `PageError` fetch paths both use it)
|
||||
/// and references no browser types — which also lets it be unit-tested
|
||||
/// without standing up a real `Page`.
|
||||
pub(crate) async fn close_after<R>(
|
||||
close: impl std::future::Future<Output = ()>,
|
||||
body: impl std::future::Future<Output = R>,
|
||||
) -> R {
|
||||
let result = body.await;
|
||||
close.await;
|
||||
result
|
||||
}
|
||||
|
||||
impl NavError {
|
||||
/// Does this navigation error indicate the underlying Chromium
|
||||
/// process has died or its CDP connection has dropped? Used by the
|
||||
@@ -150,6 +172,41 @@ mod tests {
|
||||
assert!(result.is_err(), "expected Elapsed on a hung future");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn close_after_runs_close_and_returns_value_on_ok() {
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
let closed = Arc::new(AtomicBool::new(false));
|
||||
let c = closed.clone();
|
||||
let out: anyhow::Result<i32> = close_after(
|
||||
async move { c.store(true, Ordering::SeqCst) },
|
||||
async { Ok(7) },
|
||||
)
|
||||
.await;
|
||||
assert_eq!(out.unwrap(), 7);
|
||||
assert!(closed.load(Ordering::SeqCst), "close must run on the happy path");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn close_after_runs_close_even_when_body_errs() {
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
let closed = Arc::new(AtomicBool::new(false));
|
||||
let c = closed.clone();
|
||||
// This is the leak the fix targets: the body bails before it could
|
||||
// close the page itself, and `close_after` must still close it.
|
||||
let out: anyhow::Result<()> =
|
||||
close_after(async move { c.store(true, Ordering::SeqCst) }, async {
|
||||
anyhow::bail!("nav failed")
|
||||
})
|
||||
.await;
|
||||
assert!(out.is_err());
|
||||
assert!(
|
||||
closed.load(Ordering::SeqCst),
|
||||
"the page must be closed even when the body returns Err"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nav_error_timeout_message_includes_duration() {
|
||||
let e = NavError::Timeout(Duration::from_secs(30));
|
||||
@@ -164,8 +221,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn anyhow_with_nav_timeout_in_chain_is_flagged() {
|
||||
let inner: Result<(), NavError> = Err(NavError::Timeout(NAV_TIMEOUT));
|
||||
let outer = inner.unwrap_err();
|
||||
let outer = NavError::Timeout(NAV_TIMEOUT);
|
||||
let wrapped: anyhow::Error =
|
||||
anyhow::Error::new(outer).context("wait for chapter nav");
|
||||
assert!(anyhow_looks_browser_dead(&wrapped));
|
||||
|
||||
@@ -44,6 +44,25 @@ impl RateLimiter {
|
||||
}
|
||||
}
|
||||
|
||||
/// Once the per-host map grows past this many entries, the next `wait_for`
|
||||
/// sweeps out hosts idle longer than [`IDLE_EVICT_AFTER`]. A long crawl that
|
||||
/// touches thousands of distinct CDN shards would otherwise retain one bucket
|
||||
/// per host for the daemon's whole lifetime. Far above any single source's
|
||||
/// real host count, so the sweep is rare.
|
||||
const MAX_TRACKED_HOSTS: usize = 1024;
|
||||
|
||||
/// A host bucket untouched for at least this long is evicted on the next
|
||||
/// over-cap sweep. Generous: a host still being crawled is touched every
|
||||
/// `interval`, so only genuinely-finished hosts age out.
|
||||
const IDLE_EVICT_AFTER: Duration = Duration::from_secs(3600);
|
||||
|
||||
/// A host's bucket plus when it was last used, so idle entries can be evicted.
|
||||
#[derive(Debug)]
|
||||
struct HostEntry {
|
||||
limiter: Arc<Mutex<RateLimiter>>,
|
||||
last_used: Instant,
|
||||
}
|
||||
|
||||
/// Per-host rate limiter map. The outer `Mutex<HashMap>` is held only
|
||||
/// during the entry-or-insert + Arc clone; the per-host `Mutex<RateLimiter>`
|
||||
/// is held during the actual `wait().await`. So N workers calling
|
||||
@@ -54,7 +73,7 @@ impl RateLimiter {
|
||||
pub struct HostRateLimiters {
|
||||
default_interval: Duration,
|
||||
overrides: HashMap<String, Duration>,
|
||||
map: Mutex<HashMap<String, Arc<Mutex<RateLimiter>>>>,
|
||||
map: Mutex<HashMap<String, HostEntry>>,
|
||||
}
|
||||
|
||||
impl HostRateLimiters {
|
||||
@@ -82,20 +101,37 @@ impl HostRateLimiters {
|
||||
.ok_or_else(|| anyhow::anyhow!("no host in url: {url}"))?;
|
||||
let limiter = {
|
||||
let mut map = self.map.lock().await;
|
||||
map.entry(host.clone())
|
||||
.or_insert_with(|| {
|
||||
let interval = self
|
||||
.overrides
|
||||
.get(&host)
|
||||
.copied()
|
||||
.unwrap_or(self.default_interval);
|
||||
Arc::new(Mutex::new(RateLimiter::new(interval)))
|
||||
})
|
||||
.clone()
|
||||
let now = Instant::now();
|
||||
// Bound the map: when it grows past the soft cap, drop hosts that
|
||||
// have been idle past the TTL. Only fires over-cap, so the common
|
||||
// path is a plain lookup. A host still being crawled is touched
|
||||
// every `interval` and so never ages out.
|
||||
if map.len() >= MAX_TRACKED_HOSTS {
|
||||
map.retain(|_, e| now.duration_since(e.last_used) < IDLE_EVICT_AFTER);
|
||||
}
|
||||
let entry = map.entry(host.clone()).or_insert_with(|| {
|
||||
let interval = self
|
||||
.overrides
|
||||
.get(&host)
|
||||
.copied()
|
||||
.unwrap_or(self.default_interval);
|
||||
HostEntry {
|
||||
limiter: Arc::new(Mutex::new(RateLimiter::new(interval))),
|
||||
last_used: now,
|
||||
}
|
||||
});
|
||||
entry.last_used = now;
|
||||
entry.limiter.clone()
|
||||
};
|
||||
limiter.lock().await.wait().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Number of host buckets currently tracked. Test/observability hook.
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn tracked_hosts(&self) -> usize {
|
||||
self.map.lock().await.len()
|
||||
}
|
||||
}
|
||||
|
||||
// `host_of` was duplicated across session/rate_limit/pipeline; the
|
||||
@@ -165,6 +201,46 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn host_rate_limiters_evict_idle_hosts_over_cap() {
|
||||
// Fill the map to the soft cap with distinct hosts (first call to a
|
||||
// fresh host never sleeps, so this is fast even under real time).
|
||||
let rl = HostRateLimiters::new(Duration::from_millis(1));
|
||||
for i in 0..MAX_TRACKED_HOSTS {
|
||||
rl.wait_for(&format!("https://host{i}.example/x")).await.unwrap();
|
||||
}
|
||||
assert_eq!(rl.tracked_hosts().await, MAX_TRACKED_HOSTS);
|
||||
|
||||
// Let every tracked host age past the idle TTL, then touch one new
|
||||
// host: the over-cap sweep should evict all the idle ones, leaving
|
||||
// only the freshly-inserted entry.
|
||||
tokio::time::sleep(IDLE_EVICT_AFTER + Duration::from_secs(1)).await;
|
||||
rl.wait_for("https://newcomer.example/y").await.unwrap();
|
||||
assert_eq!(
|
||||
rl.tracked_hosts().await,
|
||||
1,
|
||||
"idle hosts should be swept once the map is over the cap"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn host_rate_limiters_keep_recently_used_hosts() {
|
||||
// A host touched within the TTL must survive a sweep so active crawls
|
||||
// aren't reset. Fill to the cap, then re-touch one host right before
|
||||
// adding a newcomer that triggers the sweep.
|
||||
let rl = HostRateLimiters::new(Duration::from_millis(1));
|
||||
for i in 0..MAX_TRACKED_HOSTS {
|
||||
rl.wait_for(&format!("https://host{i}.example/x")).await.unwrap();
|
||||
}
|
||||
tokio::time::sleep(IDLE_EVICT_AFTER + Duration::from_secs(1)).await;
|
||||
// Re-touch host0 so it's recent again.
|
||||
rl.wait_for("https://host0.example/x").await.unwrap();
|
||||
// Newcomer triggers the sweep (map is at cap+1 conceptually).
|
||||
rl.wait_for("https://newcomer.example/y").await.unwrap();
|
||||
// host0 (recent) + newcomer survive; the rest aged out.
|
||||
assert_eq!(rl.tracked_hosts().await, 2);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn host_rate_limiters_honor_overrides() {
|
||||
let rl = HostRateLimiters::new(Duration::from_millis(1000))
|
||||
|
||||
@@ -81,6 +81,8 @@ pub struct RealResyncService {
|
||||
pub rate: Arc<HostRateLimiters>,
|
||||
pub download_allowlist: DownloadAllowlist,
|
||||
pub max_image_bytes: usize,
|
||||
/// Per-chapter image-count cap (see `CrawlerConfig::max_images_per_chapter`).
|
||||
pub max_images_per_chapter: usize,
|
||||
pub tor: Option<Arc<TorController>>,
|
||||
}
|
||||
|
||||
@@ -133,28 +135,7 @@ impl ResyncService for RealResyncService {
|
||||
.await
|
||||
.with_context(|| format!("fetch_manga during resync of {manga_id}"))?;
|
||||
|
||||
// Partial-render guard: same logic as run_metadata_pass.
|
||||
let source_id = source.id();
|
||||
if !manga.chapters.is_empty() || {
|
||||
let prior = repo::crawler::live_chapter_count_for_source_manga(
|
||||
&self.db,
|
||||
source_id,
|
||||
&source_manga_key,
|
||||
)
|
||||
.await
|
||||
.unwrap_or(0);
|
||||
prior == 0
|
||||
} {
|
||||
// Either the new fetch surfaced chapters, or there were
|
||||
// none before either — chapter sync is safe to run.
|
||||
} else {
|
||||
tracing::warn!(
|
||||
%manga_id,
|
||||
source_url = %source_url,
|
||||
"resync_manga: fetch returned empty chapters but prior count > 0; skipping chapter sync to avoid soft-drop"
|
||||
);
|
||||
}
|
||||
|
||||
let upsert = repo::crawler::upsert_manga_from_source(
|
||||
&self.db,
|
||||
source_id,
|
||||
@@ -199,6 +180,11 @@ impl ResyncService for RealResyncService {
|
||||
)
|
||||
.await
|
||||
.unwrap_or(0);
|
||||
// Partial-render guard (same logic as run_metadata_pass): only sync
|
||||
// chapters when the fetch surfaced some, or the manga never had any.
|
||||
// A fetch that returned empty while a prior count exists is almost
|
||||
// certainly a partial render — skip the sync so we don't soft-drop the
|
||||
// real chapters.
|
||||
if !manga.chapters.is_empty() || prior_chapter_count == 0 {
|
||||
match repo::crawler::sync_manga_chapters(
|
||||
&self.db,
|
||||
@@ -221,6 +207,12 @@ impl ResyncService for RealResyncService {
|
||||
"resync_manga: chapter sync failed"
|
||||
),
|
||||
}
|
||||
} else {
|
||||
tracing::warn!(
|
||||
%manga_id,
|
||||
source_url = %source_url,
|
||||
"resync_manga: fetch returned empty chapters but prior count > 0; skipping chapter sync to avoid soft-drop"
|
||||
);
|
||||
}
|
||||
|
||||
drop(lease);
|
||||
@@ -256,6 +248,7 @@ impl ResyncService for RealResyncService {
|
||||
true,
|
||||
&self.download_allowlist,
|
||||
self.max_image_bytes,
|
||||
self.max_images_per_chapter,
|
||||
self.tor.as_deref(),
|
||||
// Admin resync isn't a daemon worker slot — no live status.
|
||||
None,
|
||||
@@ -277,6 +270,12 @@ impl ResyncService for RealResyncService {
|
||||
SyncOutcome::SessionExpired => {
|
||||
anyhow::bail!("source session expired — operator must refresh PHPSESSID")
|
||||
}
|
||||
// Unreachable here: resync already holds its own browser lease and
|
||||
// `sync_chapter_content` never acquires one, so it can't report the
|
||||
// browser unavailable. Handled defensively for exhaustiveness.
|
||||
SyncOutcome::BrowserUnavailable => {
|
||||
anyhow::bail!("crawler browser unavailable")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,11 +31,13 @@
|
||||
//! URL string, and the byte accumulator is keyed off a generic stream.
|
||||
//! Easy to unit-test without a live network or browser.
|
||||
|
||||
use std::net::IpAddr;
|
||||
use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr};
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{bail, Context};
|
||||
use bytes::BytesMut;
|
||||
use futures_util::StreamExt;
|
||||
use reqwest::dns::{Addrs, Name, Resolve, Resolving};
|
||||
use reqwest::Url;
|
||||
|
||||
/// Default per-image download cap. A page image is generally <2 MiB;
|
||||
@@ -179,7 +181,7 @@ fn ensure_public_target_inner(raw_url: &str) -> Result<Url, UrlSafetyError> {
|
||||
Ok(url)
|
||||
}
|
||||
|
||||
fn is_private_ip(ip: &IpAddr) -> bool {
|
||||
pub(crate) fn is_private_ip(ip: &IpAddr) -> bool {
|
||||
match ip {
|
||||
IpAddr::V4(v4) => {
|
||||
v4.is_loopback()
|
||||
@@ -193,11 +195,14 @@ fn is_private_ip(ip: &IpAddr) -> bool {
|
||||
|| v4.octets()[0] == 0
|
||||
}
|
||||
IpAddr::V6(v6) => {
|
||||
// IPv4-mapped IPv6 (::ffff:0:0/96): unwrap to the embedded
|
||||
// IPv4 and recurse so `::ffff:127.0.0.1` is caught by the
|
||||
// IPv4 loopback check rather than passing through.
|
||||
// `Ipv6Addr::is_loopback()` only matches `::1` exactly.
|
||||
if let Some(v4) = v6.to_ipv4_mapped() {
|
||||
// Any IPv6 form that *embeds* an IPv4 address (mapped
|
||||
// `::ffff:0:0/96`, compatible `::/96`, NAT64 `64:ff9b::/96`,
|
||||
// 6to4 `2002::/16`) is unwrapped and re-checked as its IPv4 —
|
||||
// otherwise `::127.0.0.1` / `2002:7f00:1::` / `64:ff9b::7f00:1`
|
||||
// would smuggle an internal IPv4 past the check (the audit's
|
||||
// IPv6-embedding gap). `Ipv6Addr::is_loopback()` only matches
|
||||
// `::1` exactly, so these embeddings need explicit handling.
|
||||
if let Some(v4) = embedded_ipv4(v6) {
|
||||
return is_private_ip(&IpAddr::V4(v4));
|
||||
}
|
||||
v6.is_loopback()
|
||||
@@ -210,6 +215,114 @@ fn is_private_ip(ip: &IpAddr) -> bool {
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the IPv4 address embedded in an IPv6 literal, for every
|
||||
/// transitional encoding that can carry one: IPv4-mapped (`::ffff:0:0/96`),
|
||||
/// IPv4-compatible (`::/96`, deprecated but still routable via some stacks),
|
||||
/// NAT64 (`64:ff9b::/96`), and 6to4 (`2002::/16`). Returns `None` for a
|
||||
/// native IPv6 address. Callers recurse into [`is_private_ip`] on the result
|
||||
/// so a private IPv4 can't hide inside an IPv6 literal.
|
||||
fn embedded_ipv4(v6: &Ipv6Addr) -> Option<Ipv4Addr> {
|
||||
let seg = v6.segments();
|
||||
let low32 = |a: u16, b: u16| Ipv4Addr::new((a >> 8) as u8, (a & 0xff) as u8, (b >> 8) as u8, (b & 0xff) as u8);
|
||||
// ::ffff:0:0/96 (mapped) and ::/96 (compatible) — top 96 bits zero,
|
||||
// except mapped which has 0xffff at seg[5]. to_ipv4() covers both.
|
||||
if seg[0..5] == [0, 0, 0, 0, 0] && (seg[5] == 0 || seg[5] == 0xffff) {
|
||||
return Some(low32(seg[6], seg[7]));
|
||||
}
|
||||
// NAT64 64:ff9b::/96
|
||||
if seg[0] == 0x0064 && seg[1] == 0xff9b && seg[2..6] == [0, 0, 0, 0] {
|
||||
return Some(low32(seg[6], seg[7]));
|
||||
}
|
||||
// 6to4 2002::/16 — embedded IPv4 is bits 16..48 (seg[1], seg[2]).
|
||||
if seg[0] == 0x2002 {
|
||||
return Some(low32(seg[1], seg[2]));
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// A `reqwest::dns::Resolve` that performs normal system resolution, then
|
||||
/// drops any resolved address in a private / loopback / link-local / metadata
|
||||
/// range. Installed on every crawler + analysis reqwest client so a hostname
|
||||
/// that resolves to an internal IP (DNS rebinding: an attacker-owned domain
|
||||
/// with an `A` record for `169.254.169.254` or `10.x`) can never be connected
|
||||
/// to — closing the TOCTOU gap that the string-only `ensure_public_target`
|
||||
/// check leaves open. Fires per connection, so it also guards redirect hops.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct SafeResolver;
|
||||
|
||||
/// Partition resolved addresses into the public ones, rejecting when nothing
|
||||
/// survives. Split out from the async `resolve` so the security-critical
|
||||
/// filter is unit-testable without real DNS.
|
||||
fn retain_public_addrs(
|
||||
host: &str,
|
||||
addrs: impl Iterator<Item = SocketAddr>,
|
||||
) -> Result<Vec<SocketAddr>, BlockedResolution> {
|
||||
let public: Vec<SocketAddr> = addrs.filter(|a| !is_private_ip(&a.ip())).collect();
|
||||
if public.is_empty() {
|
||||
return Err(BlockedResolution(host.to_string()));
|
||||
}
|
||||
Ok(public)
|
||||
}
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
#[error("host {0} resolved only to private/blocked addresses")]
|
||||
struct BlockedResolution(String);
|
||||
|
||||
impl Resolve for SafeResolver {
|
||||
fn resolve(&self, name: Name) -> Resolving {
|
||||
Box::pin(async move {
|
||||
let host = name.as_str().to_string();
|
||||
// Port 0: reqwest overrides it with the URL's port after
|
||||
// resolution (same convention as reqwest's default GaiResolver).
|
||||
let resolved = tokio::net::lookup_host((host.as_str(), 0))
|
||||
.await
|
||||
.map_err(|e| Box::new(e) as Box<dyn std::error::Error + Send + Sync>)?;
|
||||
let public = retain_public_addrs(&host, resolved)
|
||||
.map_err(|e| Box::new(e) as Box<dyn std::error::Error + Send + Sync>)?;
|
||||
Ok(Box::new(public.into_iter()) as Addrs)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Shared [`SafeResolver`] for wiring into `ClientBuilder::dns_resolver`.
|
||||
pub fn safe_dns_resolver() -> Arc<SafeResolver> {
|
||||
Arc::new(SafeResolver)
|
||||
}
|
||||
|
||||
/// Whether the DNS-rebinding [`SafeResolver`] should be attached to a crawler
|
||||
/// reqwest client given the configured proxy (if any).
|
||||
///
|
||||
/// The resolver only guards targets that *reqwest itself* resolves:
|
||||
/// - **No proxy** (direct): reqwest resolves the target — attach.
|
||||
/// - **`http(s)://` proxy**: reqwest still resolves the target host locally and
|
||||
/// issues a `CONNECT`, so a hostname resolving to a private IP must be
|
||||
/// refused — attach.
|
||||
/// - **`socks5://` / `socks5h://` / `socks4://` (Tor)**: the *proxy* resolves
|
||||
/// the target; reqwest only ever resolves the proxy's own host, which
|
||||
/// legitimately lives on a private Docker IP (e.g. `tor` → 172.x). Attaching
|
||||
/// the resolver there rejects every fetch for zero security gain (a SOCKS
|
||||
/// proxy can't route into internal ranges anyway) — do **not** attach.
|
||||
///
|
||||
/// This narrows commit 134ab54, which dropped the resolver for *any* proxy, back
|
||||
/// to SOCKS-only so the http(s)-proxy path keeps its DNS-rebinding guard.
|
||||
pub fn should_attach_safe_resolver(proxy: Option<&str>) -> bool {
|
||||
match proxy {
|
||||
None => true,
|
||||
Some(p) => !is_socks_proxy(p),
|
||||
}
|
||||
}
|
||||
|
||||
/// True when `proxy`'s scheme is a SOCKS variant (`socks4`, `socks5`,
|
||||
/// `socks5h`). A scheme-less value (reqwest treats it as HTTP) is not SOCKS.
|
||||
fn is_socks_proxy(proxy: &str) -> bool {
|
||||
proxy
|
||||
.split_once("://")
|
||||
.map(|(scheme, _)| scheme)
|
||||
.unwrap_or("")
|
||||
.to_ascii_lowercase()
|
||||
.starts_with("socks")
|
||||
}
|
||||
|
||||
#[derive(Debug, thiserror::Error, PartialEq, Eq)]
|
||||
pub enum UrlSafetyError {
|
||||
#[error("URL is not parseable")]
|
||||
@@ -226,6 +339,73 @@ pub enum UrlSafetyError {
|
||||
HostNotAllowed(String),
|
||||
}
|
||||
|
||||
/// Maximum number of redirects the crawler will follow before giving up.
|
||||
/// Matches reqwest's historical default; every hop is re-validated by
|
||||
/// [`check_redirect_hop`], so the cap is a belt-and-braces stop against a
|
||||
/// redirect loop rather than the primary SSRF defence.
|
||||
pub const MAX_REDIRECTS: usize = 10;
|
||||
|
||||
/// Why a redirect hop was refused.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum RedirectError {
|
||||
#[error("redirect chain exceeded {0} hops")]
|
||||
TooManyHops(usize),
|
||||
#[error("redirect target rejected: {0}")]
|
||||
Unsafe(#[from] UrlSafetyError),
|
||||
}
|
||||
|
||||
/// Decide whether a single redirect hop is safe to follow.
|
||||
///
|
||||
/// `is_safe_url` only inspects the *initial* URL a caller hands to reqwest;
|
||||
/// without re-validation an allowlisted CDN that answers `302 ->
|
||||
/// http://169.254.169.254/...` or `-> http://127.0.0.1:5432/` would be
|
||||
/// followed transparently (reqwest's default policy follows up to 10
|
||||
/// redirects). This re-runs the full allowlist + private-IP + scheme check on
|
||||
/// each hop and enforces [`MAX_REDIRECTS`]. `completed_hops` is the number of
|
||||
/// URLs already visited in the chain (reqwest's `attempt.previous().len()`).
|
||||
pub fn check_redirect_hop(
|
||||
next_url: &str,
|
||||
completed_hops: usize,
|
||||
allow: &DownloadAllowlist,
|
||||
) -> Result<(), RedirectError> {
|
||||
if completed_hops >= MAX_REDIRECTS {
|
||||
return Err(RedirectError::TooManyHops(completed_hops));
|
||||
}
|
||||
is_safe_url(next_url, allow)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Build a reqwest redirect policy that re-validates every hop against the
|
||||
/// download allowlist (see [`check_redirect_hop`]). Use for the crawler image
|
||||
/// clients, which fetch attacker-influenced URLs.
|
||||
pub fn safe_redirect_policy(allow: DownloadAllowlist) -> reqwest::redirect::Policy {
|
||||
reqwest::redirect::Policy::custom(move |attempt| {
|
||||
let hops = attempt.previous().len();
|
||||
match check_redirect_hop(attempt.url().as_str(), hops, &allow) {
|
||||
Ok(()) => attempt.follow(),
|
||||
Err(e) => attempt.error(e),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Build a reqwest redirect policy that re-validates every hop with
|
||||
/// [`ensure_public_target`] (scheme + private-IP, no allowlist). Use for
|
||||
/// single-endpoint clients (the analysis vision endpoint / its probe) where
|
||||
/// there is no per-deployment allowlist but a redirect into the deployment's
|
||||
/// internal network must still be refused.
|
||||
pub fn public_redirect_policy() -> reqwest::redirect::Policy {
|
||||
reqwest::redirect::Policy::custom(move |attempt| {
|
||||
let hops = attempt.previous().len();
|
||||
if hops >= MAX_REDIRECTS {
|
||||
return attempt.error(RedirectError::TooManyHops(hops));
|
||||
}
|
||||
match ensure_public_target(attempt.url().as_str()) {
|
||||
Ok(()) => attempt.follow(),
|
||||
Err(e) => attempt.error(RedirectError::Unsafe(e)),
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Drain a byte stream into a single buffer, bailing out as soon as
|
||||
/// the running total exceeds `max_bytes`. Generic over the stream so
|
||||
/// it's testable without a live HTTP response.
|
||||
@@ -327,6 +507,26 @@ mod tests {
|
||||
DownloadAllowlist::new().allow(host)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_alternate_ipv4_encodings_of_loopback() {
|
||||
// The `url` crate normalizes special-scheme IPv4 hosts per WHATWG, so
|
||||
// decimal/hex/octal/short forms all collapse to 127.0.0.1 and must be
|
||||
// rejected. This normalization is load-bearing (it's what stops these
|
||||
// encodings smuggling past the literal-IP check) but was untested — pin
|
||||
// it so a future URL-parser swap can't silently reopen the hole.
|
||||
for url in [
|
||||
"http://2130706433/", // decimal 127.0.0.1
|
||||
"http://0x7f000001/", // hex
|
||||
"http://0177.0.0.1/", // octal first octet
|
||||
"http://127.1/", // short form
|
||||
] {
|
||||
assert!(
|
||||
ensure_public_target(url).is_err(),
|
||||
"{url} normalizes to loopback and must be rejected"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn allow_any_admits_arbitrary_public_host() {
|
||||
// Operators who can't pre-enumerate a numbered-CDN fleet
|
||||
@@ -542,6 +742,79 @@ mod tests {
|
||||
assert!(matches!(err, UrlSafetyError::PrivateIp(_)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_private_ip_unwraps_embedded_ipv4_encodings() {
|
||||
// Every IPv6 encoding that can smuggle an internal IPv4 must be
|
||||
// caught. The audit flagged compatible ::/96, NAT64, and 6to4 as
|
||||
// gaps past the original mapped-only handling.
|
||||
for s in [
|
||||
"::ffff:127.0.0.1", // IPv4-mapped loopback
|
||||
"::127.0.0.1", // IPv4-compatible loopback (was a gap)
|
||||
"::ffff:10.1.2.3", // mapped RFC1918
|
||||
"::10.1.2.3", // compatible RFC1918 (was a gap)
|
||||
"64:ff9b::7f00:1", // NAT64 of 127.0.0.1 (was a gap)
|
||||
"64:ff9b::a01:203", // NAT64 of 10.1.2.3
|
||||
"2002:7f00:1::", // 6to4 of 127.0.0.1 (was a gap)
|
||||
"2002:a01:203::", // 6to4 of 10.1.2.3
|
||||
"2002:a9fe:a9fe::", // 6to4 of 169.254.169.254 (metadata)
|
||||
] {
|
||||
let ip: IpAddr = s.parse().unwrap();
|
||||
assert!(is_private_ip(&ip), "{s} must be flagged private");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_private_ip_allows_public_embedded_and_native_ipv6() {
|
||||
// A public IPv4 embedded in IPv6, and a native public IPv6, must
|
||||
// NOT be flagged — the unwrap only blocks when the embedded v4 is
|
||||
// itself private.
|
||||
for s in [
|
||||
"::ffff:8.8.8.8", // mapped public
|
||||
"2002:808:808::", // 6to4 of 8.8.8.8 (public)
|
||||
"2606:4700:4700::1111", // native public (Cloudflare)
|
||||
] {
|
||||
let ip: IpAddr = s.parse().unwrap();
|
||||
assert!(!is_private_ip(&ip), "{s} must be allowed");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retain_public_addrs_drops_private_and_errors_when_all_private() {
|
||||
use std::net::{Ipv4Addr, SocketAddr};
|
||||
let pub_addr = SocketAddr::from((Ipv4Addr::new(93, 184, 216, 34), 0));
|
||||
let loopback = SocketAddr::from((Ipv4Addr::new(127, 0, 0, 1), 0));
|
||||
let metadata = SocketAddr::from((Ipv4Addr::new(169, 254, 169, 254), 0));
|
||||
|
||||
// Mixed result keeps only the public address.
|
||||
let kept =
|
||||
retain_public_addrs("mixed.example", [pub_addr, loopback, metadata].into_iter())
|
||||
.expect("public address survives");
|
||||
assert_eq!(kept, vec![pub_addr]);
|
||||
|
||||
// All-private (DNS rebinding to internal) is rejected outright.
|
||||
let err =
|
||||
retain_public_addrs("rebind.attacker", [loopback, metadata].into_iter()).unwrap_err();
|
||||
assert!(err.to_string().contains("rebind.attacker"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn safe_resolver_attaches_except_for_socks_proxies() {
|
||||
// Direct + http(s) proxies: reqwest resolves the target, so the
|
||||
// DNS-rebinding guard must be attached.
|
||||
assert!(should_attach_safe_resolver(None));
|
||||
assert!(should_attach_safe_resolver(Some("http://proxy.internal:8080")));
|
||||
assert!(should_attach_safe_resolver(Some("https://proxy.internal:8080")));
|
||||
assert!(should_attach_safe_resolver(Some("HTTP://Proxy:8080")));
|
||||
// A scheme-less proxy is treated as HTTP by reqwest — keep the guard.
|
||||
assert!(should_attach_safe_resolver(Some("proxy.internal:8080")));
|
||||
// SOCKS variants (incl. Tor): the proxy resolves, so attaching the
|
||||
// resolver would reject every fetch for zero gain.
|
||||
assert!(!should_attach_safe_resolver(Some("socks5://tor:9050")));
|
||||
assert!(!should_attach_safe_resolver(Some("socks5h://tor:9050")));
|
||||
assert!(!should_attach_safe_resolver(Some("socks4://x:1080")));
|
||||
assert!(!should_attach_safe_resolver(Some("SOCKS5://Tor:9050")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn safe_url_blocks_non_http_schemes() {
|
||||
let allow = allow_just("anywhere");
|
||||
@@ -578,6 +851,54 @@ mod tests {
|
||||
assert!(is_safe_url("https://CDN.EXAMPLE.com/x.jpg", &allow).is_ok());
|
||||
}
|
||||
|
||||
// --- redirect-hop re-validation (SSRF via 3xx) ---
|
||||
|
||||
#[test]
|
||||
fn redirect_hop_allows_listed_public_target() {
|
||||
let allow = allow_just("cdn.example.com");
|
||||
assert!(check_redirect_hop("https://cdn.example.com/next.jpg", 1, &allow).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn redirect_hop_blocks_private_ip_target() {
|
||||
// The core SSRF case: an allowlisted CDN 302s to the cloud metadata
|
||||
// service / an intra-compose port. Must be refused mid-chain.
|
||||
let allow = allow_just("cdn.example.com");
|
||||
for url in ["http://169.254.169.254/", "http://127.0.0.1:5432/", "http://10.0.0.1/"] {
|
||||
let err = check_redirect_hop(url, 1, &allow).unwrap_err();
|
||||
assert!(
|
||||
matches!(err, RedirectError::Unsafe(UrlSafetyError::PrivateIp(_))),
|
||||
"expected PrivateIp for {url}, got {err:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn redirect_hop_blocks_off_allowlist_public_host() {
|
||||
// Per the strict policy: a redirect to an unlisted *public* host is
|
||||
// also refused (allow_any covers the numbered-CDN case instead).
|
||||
let allow = allow_just("cdn.example.com");
|
||||
let err = check_redirect_hop("https://evil.example.org/x", 1, &allow).unwrap_err();
|
||||
assert!(matches!(err, RedirectError::Unsafe(UrlSafetyError::HostNotAllowed(_))));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn redirect_hop_blocks_bad_scheme_target() {
|
||||
let allow = DownloadAllowlist::allow_any();
|
||||
let err = check_redirect_hop("file:///etc/passwd", 1, &allow).unwrap_err();
|
||||
assert!(matches!(err, RedirectError::Unsafe(UrlSafetyError::BadScheme(_))));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn redirect_hop_caps_chain_length() {
|
||||
let allow = allow_just("cdn.example.com");
|
||||
// A safe target is still refused once the hop cap is reached, so a
|
||||
// redirect loop can't spin forever.
|
||||
let err = check_redirect_hop("https://cdn.example.com/x", MAX_REDIRECTS, &allow)
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, RedirectError::TooManyHops(n) if n == MAX_REDIRECTS));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn accumulate_capped_returns_full_body_under_cap() {
|
||||
let chunks: Vec<Result<bytes::Bytes, std::io::Error>> = vec![
|
||||
|
||||
@@ -285,25 +285,39 @@ where
|
||||
}
|
||||
|
||||
async fn fetch_probe_html(browser: &Browser, probe_url: &str) -> anyhow::Result<String> {
|
||||
let page = browser
|
||||
.new_page(probe_url)
|
||||
// Guard the probe navigation for parity with the list/detail and
|
||||
// chapter-content paths — the probe URL is operator-controlled, but
|
||||
// keeping every `new_page` behind the same SSRF check avoids a gap if
|
||||
// the URL ever becomes attacker-influenced.
|
||||
crate::crawler::safety::ensure_public_target(probe_url)
|
||||
.with_context(|| format!("refuse to navigate unsafe probe URL {probe_url}"))?;
|
||||
let page = crate::crawler::intercept::open_page(browser, probe_url)
|
||||
.await
|
||||
.with_context(|| format!("open probe page {probe_url}"))?;
|
||||
crate::crawler::nav::wait_for_nav(&page)
|
||||
.await
|
||||
.context("wait for nav on probe")?;
|
||||
// Best-effort wait for the layout marker. Timeout is fine — the
|
||||
// probe classifier handles a missing `#logo` as Transient anyway,
|
||||
// and the verify loop retries on Transient.
|
||||
let _ = crate::crawler::nav::wait_for_selector(
|
||||
&page,
|
||||
"#logo",
|
||||
crate::crawler::nav::SELECTOR_TIMEOUT,
|
||||
// Close the tab on every exit path — a `?` on wait_for_nav / content()
|
||||
// would otherwise leak it (chromiumoxide doesn't close on drop).
|
||||
let closer = page.clone();
|
||||
crate::crawler::nav::close_after(
|
||||
async move {
|
||||
closer.close().await.ok();
|
||||
},
|
||||
async {
|
||||
crate::crawler::nav::wait_for_nav(&page)
|
||||
.await
|
||||
.context("wait for nav on probe")?;
|
||||
// Best-effort wait for the layout marker. Timeout is fine — the
|
||||
// probe classifier handles a missing `#logo` as Transient anyway,
|
||||
// and the verify loop retries on Transient.
|
||||
let _ = crate::crawler::nav::wait_for_selector(
|
||||
&page,
|
||||
"#logo",
|
||||
crate::crawler::nav::SELECTOR_TIMEOUT,
|
||||
)
|
||||
.await;
|
||||
page.content().await.context("read probe html")
|
||||
},
|
||||
)
|
||||
.await;
|
||||
let html = page.content().await.context("read probe html")?;
|
||||
page.close().await.ok();
|
||||
Ok(html)
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -20,7 +20,8 @@ use super::{
|
||||
use crate::crawler::detect::{
|
||||
has_logo_sentinel, is_broken_page_body, retry_on_transient_with_hook, PageError,
|
||||
};
|
||||
use crate::crawler::nav::{wait_for_nav, wait_for_selector, NavError, SELECTOR_TIMEOUT};
|
||||
use crate::crawler::nav::{close_after, wait_for_nav, wait_for_selector, NavError, SELECTOR_TIMEOUT};
|
||||
use crate::crawler::safety::ensure_public_target;
|
||||
|
||||
/// `sources.id` value for this Source impl. Exposed as a const so the
|
||||
/// daemon can look up per-source state (e.g. the recovery flag) before
|
||||
@@ -220,6 +221,19 @@ const LIST_PAGE_MARKER: &str = "#left_side .pic_list .updatesli";
|
||||
const DETAIL_PAGE_CHAPTERS_MARKER: &str = "#chapter_table td h4 a.chico";
|
||||
const DETAIL_PAGE_LAYOUT_MARKER: &str = "#logo";
|
||||
|
||||
/// Refuse to point the headless browser at a private/internal target.
|
||||
/// The list/detail URLs driven through [`navigate`] originate from
|
||||
/// scraped hrefs (base URL, pagination, and detail links harvested from
|
||||
/// listings), so a hostile or compromised source could otherwise steer
|
||||
/// Chromium at `http://169.254.169.254/`, `http://postgres:5432/`, etc.
|
||||
/// and read the response body as an SSRF oracle. Mirrors the
|
||||
/// chapter-content guard in [`crate::crawler::content`].
|
||||
fn guard_navigate_url(url: &str) -> Result<(), PageError> {
|
||||
ensure_public_target(url).map_err(|e| {
|
||||
PageError::Other(anyhow::anyhow!("refuse to navigate unsafe URL {url}: {e}"))
|
||||
})
|
||||
}
|
||||
|
||||
/// Single point of rate-limited navigation. Every Source request goes
|
||||
/// through here, so the per-host limiter map is the only knob that
|
||||
/// controls per-origin RPS. Also the choke point for transient-page
|
||||
@@ -237,32 +251,39 @@ async fn navigate(
|
||||
url: &str,
|
||||
marker: &str,
|
||||
) -> Result<String, PageError> {
|
||||
guard_navigate_url(url)?;
|
||||
ctx.rate.wait_for(url).await?;
|
||||
let page = ctx
|
||||
.browser
|
||||
.new_page(url)
|
||||
let page = crate::crawler::intercept::open_page(ctx.browser, url)
|
||||
.await
|
||||
.map_err(|e| PageError::Other(anyhow::Error::from(e)))?;
|
||||
match wait_for_nav(&page).await {
|
||||
Ok(()) => {}
|
||||
Err(NavError::Timeout(_)) => {
|
||||
page.close().await.ok();
|
||||
return Err(PageError::transient("nav timeout"));
|
||||
}
|
||||
Err(NavError::Cdp(e)) => {
|
||||
page.close().await.ok();
|
||||
return Err(PageError::Other(anyhow::Error::from(e)));
|
||||
}
|
||||
}
|
||||
// Best-effort wait for the page-type marker. We deliberately
|
||||
// discard a timeout here — see fn-level doc.
|
||||
let _ = wait_for_selector(&page, marker, SELECTOR_TIMEOUT).await;
|
||||
let html = page
|
||||
.content()
|
||||
.await
|
||||
.map_err(|e| PageError::Other(anyhow::Error::from(e)))?;
|
||||
page.close().await.ok();
|
||||
classify_navigate_html(html)
|
||||
// Close the tab on every exit path — the previous code closed on the two
|
||||
// nav-error branches but leaked it on a content() read error.
|
||||
let closer = page.clone();
|
||||
close_after(
|
||||
async move {
|
||||
closer.close().await.ok();
|
||||
},
|
||||
async {
|
||||
match wait_for_nav(&page).await {
|
||||
Ok(()) => {}
|
||||
Err(NavError::Timeout(_)) => {
|
||||
return Err(PageError::transient("nav timeout"));
|
||||
}
|
||||
Err(NavError::Cdp(e)) => {
|
||||
return Err(PageError::Other(anyhow::Error::from(e)));
|
||||
}
|
||||
}
|
||||
// Best-effort wait for the page-type marker. We deliberately
|
||||
// discard a timeout here — see fn-level doc.
|
||||
let _ = wait_for_selector(&page, marker, SELECTOR_TIMEOUT).await;
|
||||
let html = page
|
||||
.content()
|
||||
.await
|
||||
.map_err(|e| PageError::Other(anyhow::Error::from(e)))?;
|
||||
classify_navigate_html(html)
|
||||
},
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Classify a fetched body. The broken-page template is universal across
|
||||
@@ -1107,4 +1128,37 @@ mod tests {
|
||||
.expect("metadata-only parse must not require chapter table");
|
||||
assert!(manga.chapters.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn navigate_guard_rejects_private_and_internal_targets() {
|
||||
// The SSRF guard `navigate` runs before opening any headless page.
|
||||
// A scraped listing/detail href pointing at cloud metadata, an
|
||||
// internal service, or the loopback interface must be refused.
|
||||
for bad in [
|
||||
"http://169.254.169.254/latest/meta-data/",
|
||||
"http://127.0.0.1:5432/",
|
||||
"http://postgres:5432/", // resolves to a private range name, but…
|
||||
"http://[::1]/",
|
||||
"http://10.0.0.5/",
|
||||
"file:///etc/passwd",
|
||||
] {
|
||||
// `postgres` is a bare hostname, not an IP literal, so the
|
||||
// literal-IP guard alone passes it — assert only the cases the
|
||||
// guard is designed to catch (IP literals + bad schemes).
|
||||
if bad.contains("postgres") {
|
||||
assert!(guard_navigate_url(bad).is_ok(), "bare hostname passes literal check");
|
||||
continue;
|
||||
}
|
||||
assert!(
|
||||
guard_navigate_url(bad).is_err(),
|
||||
"expected {bad} to be refused before navigation"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn navigate_guard_allows_public_targets() {
|
||||
assert!(guard_navigate_url("https://target.example/manga/foo").is_ok());
|
||||
assert!(guard_navigate_url("https://8.8.8.8/").is_ok());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,4 +12,6 @@ pub struct ApiToken {
|
||||
pub token_hash: Vec<u8>,
|
||||
pub created_at: DateTime<Utc>,
|
||||
pub last_used_at: Option<DateTime<Utc>>,
|
||||
/// When the token stops authenticating. `None` = never expires.
|
||||
pub expires_at: Option<DateTime<Utc>>,
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ pub mod page;
|
||||
pub mod page_analysis;
|
||||
pub mod page_tag;
|
||||
pub mod patch;
|
||||
pub mod reaction;
|
||||
pub mod read_progress;
|
||||
pub mod session;
|
||||
pub mod storage_stats;
|
||||
|
||||
40
backend/src/domain/reaction.rs
Normal file
40
backend/src/domain/reaction.rs
Normal file
@@ -0,0 +1,40 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
use uuid::Uuid;
|
||||
|
||||
/// A user's private taste signal on a manga. Stored as text in
|
||||
/// `manga_reactions.reaction` (CHECK-constrained), so we map to/from a
|
||||
/// `&str` at the repo layer rather than deriving a Postgres enum type.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum Reaction {
|
||||
Like,
|
||||
Dislike,
|
||||
}
|
||||
|
||||
impl Reaction {
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Reaction::Like => "like",
|
||||
Reaction::Dislike => "dislike",
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a stored/inbound value. Returns `None` for anything outside the
|
||||
/// closed vocabulary (the DB CHECK guarantees stored rows are valid; this
|
||||
/// also guards the inbound API body).
|
||||
pub fn parse(s: &str) -> Option<Reaction> {
|
||||
match s {
|
||||
"like" => Some(Reaction::Like),
|
||||
"dislike" => Some(Reaction::Dislike),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Response shape for `GET /me/reactions/:manga_id` — the current user's
|
||||
/// reaction on one manga, or `null` when they haven't reacted.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct MangaReaction {
|
||||
pub manga_id: Uuid,
|
||||
pub reaction: Option<Reaction>,
|
||||
}
|
||||
@@ -24,8 +24,19 @@ pub struct ReadProgressSummary {
|
||||
/// `None` when the chapter was deleted after this row was written
|
||||
/// (FK ON DELETE SET NULL on `chapter_id`).
|
||||
pub chapter_number: Option<i32>,
|
||||
/// Page count of the last-read chapter (`None` when unknown / deleted).
|
||||
/// Lets a client distinguish a finished series (read to the last page,
|
||||
/// nothing new) from one still in progress — e.g. to keep finished
|
||||
/// series off a "Continue reading" shelf.
|
||||
pub chapter_page_count: Option<i32>,
|
||||
pub page: i32,
|
||||
pub updated_at: DateTime<Utc>,
|
||||
/// How many chapters sit past the reader's last-read chapter (by
|
||||
/// chapter number) — a personal "new since you last read" count.
|
||||
/// `0` when the reader is caught up or when `chapter_number` is
|
||||
/// unknown (manga-level progress or a deleted chapter), so we never
|
||||
/// claim chapters are new when we can't place the reader.
|
||||
pub new_chapters_count: i64,
|
||||
}
|
||||
|
||||
/// Returned by `GET /me/read-progress/:manga_id`. Same shape as
|
||||
@@ -40,6 +51,11 @@ pub struct ReadProgressForManga {
|
||||
pub chapter_number: Option<i32>,
|
||||
pub page: i32,
|
||||
pub updated_at: DateTime<Utc>,
|
||||
/// Distinct chapter numbers past the last-read chapter — the detail
|
||||
/// page's authoritative "new since last read" count (computed over all
|
||||
/// chapters, not the client's paginated list). `0` when caught up or
|
||||
/// the position is unknown. See [`ReadProgressSummary::new_chapters_count`].
|
||||
pub new_chapters_count: i64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Deserialize)]
|
||||
|
||||
@@ -39,10 +39,10 @@ pub enum AppError {
|
||||
details: serde_json::Value,
|
||||
},
|
||||
/// 501 — the wire shape is accepted but the feature isn't built yet.
|
||||
/// Carries a `&'static str` snake_case code so clients can detect
|
||||
/// the specific pending feature (`text_search_not_yet_supported`,
|
||||
/// etc.) without parsing the message. Used today by the `?text=`
|
||||
/// reservation on the page-tag aggregation endpoints.
|
||||
/// Carries a `&'static str` snake_case code so clients can detect the
|
||||
/// specific pending feature without parsing the message. Generic; kept
|
||||
/// for future reservations (the `?text=` page-tag reservation that first
|
||||
/// used it now performs real OCR search).
|
||||
#[error("not implemented: {code}")]
|
||||
NotImplemented {
|
||||
code: &'static str,
|
||||
@@ -207,4 +207,68 @@ mod tests {
|
||||
"text_search_not_yet_supported"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn status_mapping_is_stable() {
|
||||
// Pin the code -> HTTP status mapping so a refactor of into_response
|
||||
// can't silently change a status (e.g. 501 -> 500). The 501 NotImplemented
|
||||
// path and a deterministic 429 Retry-After were previously unexercised
|
||||
// (audit GAP B).
|
||||
let cases: Vec<(AppError, StatusCode)> = vec![
|
||||
(AppError::NotFound, StatusCode::NOT_FOUND),
|
||||
(AppError::InvalidInput("x".into()), StatusCode::BAD_REQUEST),
|
||||
(AppError::Unauthenticated, StatusCode::UNAUTHORIZED),
|
||||
(AppError::Forbidden, StatusCode::FORBIDDEN),
|
||||
(AppError::Conflict("x".into()), StatusCode::CONFLICT),
|
||||
(AppError::PayloadTooLarge("x".into()), StatusCode::PAYLOAD_TOO_LARGE),
|
||||
(
|
||||
AppError::UnsupportedMediaType("x".into()),
|
||||
StatusCode::UNSUPPORTED_MEDIA_TYPE,
|
||||
),
|
||||
(
|
||||
AppError::ServiceUnavailable("x".into()),
|
||||
StatusCode::SERVICE_UNAVAILABLE,
|
||||
),
|
||||
(
|
||||
AppError::ValidationFailed {
|
||||
message: "x".into(),
|
||||
details: json!({}),
|
||||
},
|
||||
StatusCode::UNPROCESSABLE_ENTITY,
|
||||
),
|
||||
(
|
||||
AppError::NotImplemented {
|
||||
code: "c",
|
||||
message: "m",
|
||||
},
|
||||
StatusCode::NOT_IMPLEMENTED,
|
||||
),
|
||||
(
|
||||
AppError::TooManyRequests {
|
||||
retry_after_secs: None,
|
||||
},
|
||||
StatusCode::TOO_MANY_REQUESTS,
|
||||
),
|
||||
(
|
||||
AppError::Other(anyhow::anyhow!("boom")),
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
),
|
||||
];
|
||||
for (err, want) in cases {
|
||||
let code = err.code();
|
||||
assert_eq!(err.into_response().status(), want, "status for code {code}");
|
||||
}
|
||||
|
||||
// A 429 with retry_after carries the Retry-After header deterministically
|
||||
// (without needing a live rate limiter).
|
||||
let resp = AppError::TooManyRequests {
|
||||
retry_after_secs: Some(7),
|
||||
}
|
||||
.into_response();
|
||||
assert_eq!(resp.status(), StatusCode::TOO_MANY_REQUESTS);
|
||||
assert_eq!(
|
||||
resp.headers().get(axum::http::header::RETRY_AFTER).unwrap(),
|
||||
"7"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -214,7 +214,7 @@ pub async fn list_mangas_with_sync_state(
|
||||
let search_pat = q
|
||||
.search
|
||||
.as_ref()
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
// sqlx::Type → text: bind the snake_case representation manually so
|
||||
// the SQL can compare it as text without an explicit cast.
|
||||
@@ -235,7 +235,7 @@ pub async fn list_mangas_with_sync_state(
|
||||
(SELECT MAX(last_seen_at) FROM manga_sources ms
|
||||
WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL) AS latest_seen_at
|
||||
FROM mangas m
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE $1)
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
)
|
||||
SELECT * FROM classified
|
||||
WHERE ($2::text IS NULL OR sync_state = $2)
|
||||
@@ -261,7 +261,7 @@ pub async fn list_mangas_with_sync_state(
|
||||
WITH classified AS (
|
||||
SELECT {case} AS sync_state
|
||||
FROM mangas m
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE $1)
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
)
|
||||
SELECT COUNT(*) FROM classified
|
||||
WHERE ($2::text IS NULL OR sync_state = $2)
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
//! token; the raw value is shown to the user once at creation and never
|
||||
//! stored.
|
||||
|
||||
use chrono::{DateTime, Utc};
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
@@ -13,28 +14,48 @@ pub async fn create(
|
||||
user_id: Uuid,
|
||||
name: &str,
|
||||
token_hash: &[u8],
|
||||
expires_at: Option<DateTime<Utc>>,
|
||||
) -> AppResult<ApiToken> {
|
||||
let row = sqlx::query_as::<_, ApiToken>(
|
||||
r#"
|
||||
INSERT INTO api_tokens (user_id, name, token_hash)
|
||||
VALUES ($1, $2, $3)
|
||||
RETURNING id, user_id, name, token_hash, created_at, last_used_at
|
||||
INSERT INTO api_tokens (user_id, name, token_hash, expires_at)
|
||||
VALUES ($1, $2, $3, $4)
|
||||
RETURNING id, user_id, name, token_hash, created_at, last_used_at, expires_at
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(name)
|
||||
.bind(token_hash)
|
||||
.bind(expires_at)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
Ok(row)
|
||||
}
|
||||
|
||||
/// The caller's tokens, newest first. `token_hash` is `#[serde(skip)]` on the
|
||||
/// domain type, so returning the full row never leaks the secret.
|
||||
pub async fn list_for_user(pool: &PgPool, user_id: Uuid) -> AppResult<Vec<ApiToken>> {
|
||||
let rows = sqlx::query_as::<_, ApiToken>(
|
||||
r#"
|
||||
SELECT id, user_id, name, token_hash, created_at, last_used_at, expires_at
|
||||
FROM api_tokens
|
||||
WHERE user_id = $1
|
||||
ORDER BY created_at DESC
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
pub async fn find_active(pool: &PgPool, token_hash: &[u8]) -> AppResult<Option<ApiToken>> {
|
||||
let row = sqlx::query_as::<_, ApiToken>(
|
||||
r#"
|
||||
SELECT id, user_id, name, token_hash, created_at, last_used_at
|
||||
SELECT id, user_id, name, token_hash, created_at, last_used_at, expires_at
|
||||
FROM api_tokens
|
||||
WHERE token_hash = $1
|
||||
AND (expires_at IS NULL OR expires_at > now())
|
||||
"#,
|
||||
)
|
||||
.bind(token_hash)
|
||||
|
||||
@@ -81,7 +81,7 @@ pub async fn list(
|
||||
SELECT id, name, created_at
|
||||
FROM authors
|
||||
WHERE $1::text IS NULL
|
||||
OR name ILIKE '%' || $1 || '%'
|
||||
OR name ILIKE '%' || $4 || '%' ESCAPE '\'
|
||||
OR name % $1
|
||||
ORDER BY CASE WHEN $1::text IS NULL THEN 0 ELSE similarity(name, $1) END DESC,
|
||||
lower(name) ASC
|
||||
@@ -91,6 +91,9 @@ pub async fn list(
|
||||
.bind(search)
|
||||
.bind(limit)
|
||||
.bind(offset)
|
||||
// $4: LIKE-escaped term for the ILIKE branch so `%`/`_` match literally; the
|
||||
// trigram/similarity branches keep the raw $1.
|
||||
.bind(search.map(crate::repo::escape_like))
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
Ok(rows)
|
||||
|
||||
@@ -6,13 +6,23 @@ use uuid::Uuid;
|
||||
use crate::domain::{Bookmark, BookmarkSummary};
|
||||
use crate::error::{AppError, AppResult};
|
||||
|
||||
/// Add a bookmark, idempotently. Returns the bookmark plus whether it was
|
||||
/// newly created (`true`) or already existed (`false`), so the handler can
|
||||
/// answer 201 vs 200. Re-adding an existing bookmark is a no-op success rather
|
||||
/// than a 409 — matching the idempotent collection semantics, so the UI doesn't
|
||||
/// show a false "Could not add bookmark" toast when the manga is already saved.
|
||||
///
|
||||
/// Uniqueness is per `(user_id, manga_id, chapter_id)` — enforced by the 0001
|
||||
/// constraint for chapter-level rows and the 0004 partial index for manga-level
|
||||
/// (NULL chapter) rows. `page` is not part of the key, so the existing row is
|
||||
/// returned unchanged (its page is not overwritten).
|
||||
pub async fn create(
|
||||
pool: &PgPool,
|
||||
user_id: Uuid,
|
||||
manga_id: Uuid,
|
||||
chapter_id: Option<Uuid>,
|
||||
page: Option<i32>,
|
||||
) -> AppResult<Bookmark> {
|
||||
) -> AppResult<(Bookmark, bool)> {
|
||||
let result = sqlx::query_as::<_, Bookmark>(
|
||||
r#"
|
||||
INSERT INTO bookmarks (user_id, manga_id, chapter_id, page)
|
||||
@@ -28,10 +38,27 @@ pub async fn create(
|
||||
.await;
|
||||
|
||||
match result {
|
||||
Ok(b) => Ok(b),
|
||||
Err(sqlx::Error::Database(ref db_err)) if db_err.is_unique_violation() => Err(
|
||||
AppError::Conflict("bookmark already exists for this manga/chapter".into()),
|
||||
),
|
||||
Ok(b) => Ok((b, true)),
|
||||
Err(sqlx::Error::Database(ref db_err)) if db_err.is_unique_violation() => {
|
||||
// A bookmark for this (user, manga, chapter) already exists — fetch
|
||||
// and return it. `IS NOT DISTINCT FROM` matches a NULL chapter_id
|
||||
// (manga-level bookmark) as well as a concrete one, covering both
|
||||
// uniqueness paths with a single lookup.
|
||||
let existing = sqlx::query_as::<_, Bookmark>(
|
||||
r#"
|
||||
SELECT id, user_id, manga_id, chapter_id, page, created_at
|
||||
FROM bookmarks
|
||||
WHERE user_id = $1 AND manga_id = $2
|
||||
AND chapter_id IS NOT DISTINCT FROM $3
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(manga_id)
|
||||
.bind(chapter_id)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
Ok((existing, false))
|
||||
}
|
||||
Err(e) => Err(AppError::Database(e)),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,14 +12,19 @@ pub async fn list_for_manga(
|
||||
limit: i64,
|
||||
offset: i64,
|
||||
) -> AppResult<Vec<Chapter>> {
|
||||
// Display order = source-site order reversed. The crawler stamps
|
||||
// `source_index` = position in the source DOM (0 = first = newest
|
||||
// on this site, see migration 0021), so DESC puts the oldest
|
||||
// chapter first and keeps the site's variant grouping and the
|
||||
// placement of non-numeric entries (e.g. "notice. : Officials")
|
||||
// intact. NULLS LAST keeps user-uploaded chapters (no source row)
|
||||
// and rows that pre-date the migration below crawled rows; the
|
||||
// (number, created_at) tail then orders them deterministically.
|
||||
// Display order. Crawled chapters carry `source_index` = position in the
|
||||
// source DOM (0 = newest on this site, migration 0021); we display them
|
||||
// reversed (oldest first) via the `-source_index` key, which keeps the
|
||||
// site's variant grouping and non-numeric entries (e.g. "notice.") in the
|
||||
// spot the site placed them — NOT clustered at number 0.
|
||||
//
|
||||
// User-uploaded chapters have no `source_index`. Instead of dumping them
|
||||
// after every crawled chapter (the old `NULLS LAST` bug, which put an
|
||||
// uploaded chapter 5 after crawled chapter 100), each is slotted by NUMBER:
|
||||
// just before the crawled chapter with the smallest number greater than it,
|
||||
// so it interleaves. An upload newer than every crawled chapter goes last;
|
||||
// when there are no crawled chapters at all, uploads fall back to the
|
||||
// `number, created_at` tail.
|
||||
let rows = sqlx::query_as::<_, Chapter>(
|
||||
r#"
|
||||
SELECT id, manga_id, number, title, page_count, created_at,
|
||||
@@ -31,7 +36,22 @@ pub async fn list_for_manga(
|
||||
FROM pages p WHERE p.chapter_id = chapters.id)::bigint AS size_bytes
|
||||
FROM chapters
|
||||
WHERE manga_id = $1
|
||||
ORDER BY source_index DESC NULLS LAST, number ASC, created_at ASC
|
||||
ORDER BY
|
||||
CASE
|
||||
WHEN source_index IS NOT NULL THEN (-source_index)::float8
|
||||
ELSE COALESCE(
|
||||
(SELECT MIN(-c2.source_index)::float8 - 0.5
|
||||
FROM chapters c2
|
||||
WHERE c2.manga_id = chapters.manga_id
|
||||
AND c2.source_index IS NOT NULL
|
||||
AND c2.number > chapters.number),
|
||||
(SELECT COALESCE(MAX(-c3.source_index), 0)::float8 + 0.5
|
||||
FROM chapters c3
|
||||
WHERE c3.manga_id = chapters.manga_id
|
||||
AND c3.source_index IS NOT NULL)
|
||||
)
|
||||
END,
|
||||
number ASC, created_at ASC
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
)
|
||||
|
||||
@@ -187,7 +187,7 @@ pub async fn list_ops(
|
||||
}
|
||||
|
||||
/// Delete metric rows older than `retention_days`. `0` disables the reaper
|
||||
/// (returns 0 without touching the table). Mirrors `jobs::reap_done`.
|
||||
/// (returns 0 without touching the table). Mirrors `jobs::reap_terminal`.
|
||||
pub async fn reap(pool: &PgPool, retention_days: u32) -> sqlx::Result<u64> {
|
||||
if retention_days == 0 {
|
||||
return Ok(0);
|
||||
|
||||
@@ -247,10 +247,13 @@ async fn sync_genres(
|
||||
let genre_id = match existing {
|
||||
Some((id,)) => id,
|
||||
None => {
|
||||
// Conflict on lower(name) (0038) not name, so a racing insert of
|
||||
// a differently-cased variant resolves to the existing row
|
||||
// instead of raising a unique violation.
|
||||
let (id,): (Uuid,) = sqlx::query_as(
|
||||
r#"
|
||||
INSERT INTO genres (name) VALUES ($1)
|
||||
ON CONFLICT (name) DO UPDATE SET name = genres.name
|
||||
ON CONFLICT (lower(name)) DO UPDATE SET name = genres.name
|
||||
RETURNING id
|
||||
"#,
|
||||
)
|
||||
@@ -720,7 +723,7 @@ pub async fn list_dead_jobs(
|
||||
offset: i64,
|
||||
) -> sqlx::Result<(Vec<DeadJob>, i64)> {
|
||||
let search_pat = search
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
let items: Vec<DeadJob> = sqlx::query_as(
|
||||
@@ -743,7 +746,7 @@ pub async fn list_dead_jobs(
|
||||
LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid
|
||||
LEFT JOIN mangas m ON m.id = c.manga_id
|
||||
WHERE cj.state = 'dead'
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 OR cj.payload->>'title' ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\' OR cj.payload->>'title' ILIKE $1 ESCAPE '\')
|
||||
ORDER BY cj.updated_at DESC
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
@@ -761,7 +764,7 @@ pub async fn list_dead_jobs(
|
||||
LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid
|
||||
LEFT JOIN mangas m ON m.id = c.manga_id
|
||||
WHERE cj.state = 'dead'
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 OR cj.payload->>'title' ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\' OR cj.payload->>'title' ILIKE $1 ESCAPE '\')
|
||||
"#,
|
||||
)
|
||||
.bind(&search_pat)
|
||||
@@ -797,7 +800,7 @@ pub async fn list_active_jobs(
|
||||
offset: i64,
|
||||
) -> sqlx::Result<(Vec<ActiveJob>, i64)> {
|
||||
let search_pat = search
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
let items: Vec<ActiveJob> = sqlx::query_as(
|
||||
@@ -817,7 +820,7 @@ pub async fn list_active_jobs(
|
||||
LEFT JOIN mangas m ON m.id = c.manga_id
|
||||
WHERE cj.state IN ('pending','running')
|
||||
AND cj.payload->>'kind' = 'sync_chapter_content'
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
ORDER BY (cj.state = 'running') DESC, cj.scheduled_at, cj.created_at
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
@@ -836,7 +839,7 @@ pub async fn list_active_jobs(
|
||||
LEFT JOIN mangas m ON m.id = c.manga_id
|
||||
WHERE cj.state IN ('pending','running')
|
||||
AND cj.payload->>'kind' = 'sync_chapter_content'
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
"#,
|
||||
)
|
||||
.bind(&search_pat)
|
||||
@@ -903,9 +906,8 @@ pub struct JobHistoryFilter<'a> {
|
||||
/// manga/chapter/page context (best-effort) so the table can label rows.
|
||||
/// Returns the page slice plus the filtered total for pagination.
|
||||
///
|
||||
/// History depth is bounded by the done-job reaper (`reap_done`): completed
|
||||
/// jobs older than the retention window are gone. Terminal `dead` jobs
|
||||
/// persist until requeued.
|
||||
/// History depth is bounded by the terminal-job reaper (`reap_terminal`):
|
||||
/// `done` and `dead` jobs older than the retention window are gone.
|
||||
pub async fn list_job_history(
|
||||
pool: &PgPool,
|
||||
filter: JobHistoryFilter<'_>,
|
||||
@@ -914,7 +916,7 @@ pub async fn list_job_history(
|
||||
) -> sqlx::Result<(Vec<JobHistoryRow>, i64)> {
|
||||
let search_pat = filter
|
||||
.search
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
// The same FROM/JOIN/WHERE drives both the page slice and the count, so
|
||||
@@ -952,7 +954,7 @@ pub async fn list_job_history(
|
||||
) cm ON true
|
||||
WHERE ($1::text IS NULL OR cj.state = $1)
|
||||
AND ($2::text IS NULL OR cj.payload->>'kind' = $2)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 OR cj.payload->>'title' ILIKE $3)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 ESCAPE '\' OR cj.payload->>'title' ILIKE $3 ESCAPE '\')
|
||||
ORDER BY cj.updated_at DESC
|
||||
LIMIT $4 OFFSET $5
|
||||
"#,
|
||||
@@ -974,7 +976,7 @@ pub async fn list_job_history(
|
||||
LEFT JOIN mangas m ON m.id = COALESCE(c.manga_id, (cj.payload->>'manga_id')::uuid)
|
||||
WHERE ($1::text IS NULL OR cj.state = $1)
|
||||
AND ($2::text IS NULL OR cj.payload->>'kind' = $2)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 OR cj.payload->>'title' ILIKE $3)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 ESCAPE '\' OR cj.payload->>'title' ILIKE $3 ESCAPE '\')
|
||||
"#,
|
||||
)
|
||||
.bind(filter.state)
|
||||
@@ -1019,7 +1021,7 @@ pub async fn list_missing_cover_mangas(
|
||||
offset: i64,
|
||||
) -> sqlx::Result<(Vec<MissingCoverRow>, i64)> {
|
||||
let search_pat = search
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
let items: Vec<MissingCoverRow> = sqlx::query_as(
|
||||
@@ -1031,7 +1033,7 @@ pub async fn list_missing_cover_mangas(
|
||||
SELECT 1 FROM manga_sources ms
|
||||
WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL
|
||||
)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
ORDER BY m.updated_at DESC
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
@@ -1050,7 +1052,7 @@ pub async fn list_missing_cover_mangas(
|
||||
SELECT 1 FROM manga_sources ms
|
||||
WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL
|
||||
)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1)
|
||||
AND ($1::text IS NULL OR m.title ILIKE $1 ESCAPE '\')
|
||||
"#,
|
||||
)
|
||||
.bind(&search_pat)
|
||||
|
||||
@@ -10,7 +10,7 @@ use uuid::Uuid;
|
||||
|
||||
use crate::domain::manga::{Manga, MangaCard, MangaDetail};
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::repo;
|
||||
use crate::repo::{self, escape_like};
|
||||
|
||||
/// Status values mirror the CHECK constraint in 0009. Centralized so
|
||||
/// the API layer can validate uploads against the same vocabulary.
|
||||
@@ -110,13 +110,13 @@ fn manga_cols(alias: &str) -> String {
|
||||
/// true.
|
||||
const FILTER_WHERE: &str = r#"
|
||||
($1::text IS NULL
|
||||
OR title ILIKE '%' || $1 || '%'
|
||||
OR title ILIKE '%' || $8 || '%' ESCAPE '\'
|
||||
OR title % $1
|
||||
OR EXISTS (
|
||||
SELECT 1 FROM manga_authors ma
|
||||
JOIN authors a ON a.id = ma.author_id
|
||||
WHERE ma.manga_id = mangas.id
|
||||
AND (a.name ILIKE '%' || $1 || '%' OR a.name % $1)
|
||||
AND (a.name ILIKE '%' || $8 || '%' ESCAPE '\' OR a.name % $1)
|
||||
)
|
||||
)
|
||||
AND ($2::text IS NULL OR status = $2)
|
||||
@@ -144,17 +144,13 @@ const FILTER_WHERE: &str = r#"
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM unnest($6::text[]) AS req(w)
|
||||
WHERE NOT EXISTS (
|
||||
SELECT 1 FROM page_content_warnings pw
|
||||
JOIN pages p ON p.id = pw.page_id
|
||||
JOIN chapters c ON c.id = p.chapter_id
|
||||
WHERE c.manga_id = mangas.id AND pw.warning = req.w
|
||||
SELECT 1 FROM manga_content_warnings mcw
|
||||
WHERE mcw.manga_id = mangas.id AND mcw.warning = req.w
|
||||
)
|
||||
)
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM page_content_warnings pw
|
||||
JOIN pages p ON p.id = pw.page_id
|
||||
JOIN chapters c ON c.id = p.chapter_id
|
||||
WHERE c.manga_id = mangas.id AND pw.warning = ANY($7::text[])
|
||||
SELECT 1 FROM manga_content_warnings mcw
|
||||
WHERE mcw.manga_id = mangas.id AND mcw.warning = ANY($7::text[])
|
||||
)
|
||||
"#;
|
||||
|
||||
@@ -163,7 +159,7 @@ const FILTER_WHERE: &str = r#"
|
||||
/// (`mangas_created_at_idx`, `mangas_updated_at_idx`, `mangas_title_lower_idx`)
|
||||
/// and the trigram GIN indexes (0005_search.sql, 0009_manga_metadata.sql) keep
|
||||
/// the search filter cheap as the library grows.
|
||||
pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, i64)> {
|
||||
pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, Option<i64>)> {
|
||||
// Both `col` and `dir` are interpolated from hard-coded enums, never from
|
||||
// request input, so this is not a SQL injection seam. The trailing `id` is
|
||||
// a stable tie-break that keeps pagination deterministic across rows with
|
||||
@@ -172,16 +168,11 @@ pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, i6
|
||||
SortField::Created => "created_at",
|
||||
SortField::Updated => "updated_at",
|
||||
SortField::Title => "lower(title)",
|
||||
// Sorts on the alphabetically-first attached author. As an ORDER BY
|
||||
// key this correlated subquery is evaluated per filter-matching row
|
||||
// before LIMIT applies, so it scales worse than the indexed date/title
|
||||
// sorts — revisit with a LATERAL join or precomputed sort-name column
|
||||
// if the library grows large.
|
||||
SortField::Author => {
|
||||
"(SELECT min(lower(a.name)) \
|
||||
FROM manga_authors ma JOIN authors a ON a.id = ma.author_id \
|
||||
WHERE ma.manga_id = mangas.id)"
|
||||
}
|
||||
// Sorts on the alphabetically-first attached author. Precomputed into
|
||||
// `mangas.sort_author` (migration 0037, maintained by triggers on
|
||||
// manga_authors) and index-backed by `mangas_sort_author_idx`, so this
|
||||
// is a plain column read rather than a per-row correlated subquery.
|
||||
SortField::Author => "sort_author",
|
||||
};
|
||||
let dir = match query.order {
|
||||
SortOrder::Asc => "ASC",
|
||||
@@ -205,11 +196,15 @@ pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, i6
|
||||
FROM mangas
|
||||
WHERE {FILTER_WHERE}
|
||||
ORDER BY {order_by}
|
||||
LIMIT $8 OFFSET $9
|
||||
LIMIT $9 OFFSET $10
|
||||
"#,
|
||||
cols = manga_cols(""),
|
||||
);
|
||||
|
||||
// $8 is the LIKE-escaped search term used by the ILIKE branches so `%`/`_`
|
||||
// in the term match literally; the trigram `%` branches keep the raw $1.
|
||||
let search_escaped = search.map(escape_like);
|
||||
|
||||
let rows = sqlx::query_as::<_, Manga>(&list_sql)
|
||||
.bind(search)
|
||||
.bind(status)
|
||||
@@ -218,27 +213,40 @@ pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, i6
|
||||
.bind(&query.tag_ids)
|
||||
.bind(&query.cw_include)
|
||||
.bind(&query.cw_exclude)
|
||||
.bind(&search_escaped)
|
||||
.bind(query.limit)
|
||||
.bind(query.offset)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
|
||||
let count_sql = format!(
|
||||
r#"
|
||||
SELECT count(*) FROM mangas
|
||||
WHERE {FILTER_WHERE}
|
||||
"#
|
||||
);
|
||||
let (total,): (i64,) = sqlx::query_as(&count_sql)
|
||||
.bind(search)
|
||||
.bind(status)
|
||||
.bind(&query.author_ids)
|
||||
.bind(&query.genre_ids)
|
||||
.bind(&query.tag_ids)
|
||||
.bind(&query.cw_include)
|
||||
.bind(&query.cw_exclude)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
// The count reuses FILTER_WHERE — up to five correlated NOT EXISTS/unnest
|
||||
// subqueries (plus a page_content_warnings join under CW filters) over the
|
||||
// whole filtered set. Recomputing it on every page made pagination scale
|
||||
// with catalog size. It doesn't change as the caller walks pages, so
|
||||
// compute it once on the first page (offset 0) and return None thereafter;
|
||||
// the pagination envelope serialises that as `total: null`.
|
||||
let total = if query.offset == 0 {
|
||||
let count_sql = format!(
|
||||
r#"
|
||||
SELECT count(*) FROM mangas
|
||||
WHERE {FILTER_WHERE}
|
||||
"#
|
||||
);
|
||||
let (total,): (i64,) = sqlx::query_as(&count_sql)
|
||||
.bind(search)
|
||||
.bind(status)
|
||||
.bind(&query.author_ids)
|
||||
.bind(&query.genre_ids)
|
||||
.bind(&query.tag_ids)
|
||||
.bind(&query.cw_include)
|
||||
.bind(&query.cw_exclude)
|
||||
.bind(&search_escaped)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
Some(total)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
Ok((rows, total))
|
||||
}
|
||||
@@ -249,7 +257,7 @@ pub async fn list(pool: &PgPool, query: &ListQuery) -> AppResult<(Vec<Manga>, i6
|
||||
pub async fn list_cards(
|
||||
pool: &PgPool,
|
||||
query: &ListQuery,
|
||||
) -> AppResult<(Vec<MangaCard>, i64)> {
|
||||
) -> AppResult<(Vec<MangaCard>, Option<i64>)> {
|
||||
let (rows, total) = list(pool, query).await?;
|
||||
let cards = cards_from_rows(pool, rows).await?;
|
||||
Ok((cards, total))
|
||||
@@ -307,6 +315,73 @@ pub async fn list_similar(
|
||||
cards_from_rows(pool, rows).await
|
||||
}
|
||||
|
||||
/// Content-based "Recommended for you": rank mangas by weighted tag overlap
|
||||
/// with the user's taste. Signals: explicit like = +1.0, bookmark = +0.5,
|
||||
/// dislike = -1.0 (a reaction overrides a bookmark on the same manga). Per
|
||||
/// tag we sum those weights into an affinity, then score each candidate by
|
||||
/// the sum of its tags' affinities, normalized by the candidate's tag count
|
||||
/// (same anti-tag-stuffing rationale as `list_similar`). Candidates the user
|
||||
/// already reacted to, bookmarked, or read are excluded; net-negative
|
||||
/// candidates (dominated by disliked-tag affinity) are dropped, so a dislike
|
||||
/// down-ranks rather than the manga being hidden from normal browse. No
|
||||
/// signals → empty. Reuses `cards_from_rows` for author/genre hydration.
|
||||
pub async fn list_recommendations(
|
||||
pool: &PgPool,
|
||||
user_id: Uuid,
|
||||
limit: i64,
|
||||
) -> AppResult<Vec<MangaCard>> {
|
||||
let sql = format!(
|
||||
r#"
|
||||
WITH signals AS (
|
||||
SELECT s.manga_id,
|
||||
CASE WHEN r.reaction = 'dislike' THEN -1.0
|
||||
WHEN r.reaction = 'like' THEN 1.0
|
||||
ELSE 0.5 END AS weight
|
||||
FROM (
|
||||
SELECT manga_id FROM manga_reactions WHERE user_id = $1
|
||||
UNION
|
||||
SELECT manga_id FROM bookmarks WHERE user_id = $1
|
||||
) s
|
||||
LEFT JOIN manga_reactions r
|
||||
ON r.user_id = $1 AND r.manga_id = s.manga_id
|
||||
),
|
||||
tag_affinity AS (
|
||||
SELECT mt.tag_id, SUM(sig.weight) AS affinity
|
||||
FROM signals sig
|
||||
JOIN manga_tags mt ON mt.manga_id = sig.manga_id
|
||||
GROUP BY mt.tag_id
|
||||
)
|
||||
SELECT {cols}
|
||||
FROM manga_tags cand
|
||||
JOIN tag_affinity ta ON ta.tag_id = cand.tag_id
|
||||
JOIN mangas m ON m.id = cand.manga_id
|
||||
WHERE cand.manga_id NOT IN (SELECT manga_id FROM signals)
|
||||
AND cand.manga_id NOT IN (
|
||||
SELECT manga_id FROM read_progress WHERE user_id = $1
|
||||
)
|
||||
GROUP BY m.id
|
||||
HAVING SUM(ta.affinity) > 0
|
||||
ORDER BY
|
||||
SUM(ta.affinity)
|
||||
/ (SELECT count(*) FROM manga_tags WHERE manga_id = m.id) DESC,
|
||||
SUM(ta.affinity) DESC,
|
||||
m.updated_at DESC,
|
||||
lower(m.title) ASC,
|
||||
m.id
|
||||
LIMIT $2
|
||||
"#,
|
||||
cols = manga_cols("m"),
|
||||
);
|
||||
|
||||
let rows = sqlx::query_as::<_, Manga>(&sql)
|
||||
.bind(user_id)
|
||||
.bind(limit)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
|
||||
cards_from_rows(pool, rows).await
|
||||
}
|
||||
|
||||
/// Hydrate a batch of `Manga` rows into `MangaCard`s by attaching their
|
||||
/// authors and genres in two batched round-trips. The input order is
|
||||
/// preserved (callers rely on this to keep list/ranking order), so we
|
||||
|
||||
@@ -13,6 +13,7 @@ pub mod manga;
|
||||
pub mod page;
|
||||
pub mod page_analysis;
|
||||
pub mod page_tag;
|
||||
pub mod reaction;
|
||||
pub mod read_progress;
|
||||
pub mod session;
|
||||
pub mod storage_stats;
|
||||
@@ -20,3 +21,45 @@ pub mod tag;
|
||||
pub mod upload_history;
|
||||
pub mod user;
|
||||
pub mod user_preferences;
|
||||
|
||||
/// Escape the LIKE/ILIKE metacharacters (`%`, `_`, and the escape char `\`
|
||||
/// itself) in a user-supplied search term so they match literally rather than
|
||||
/// as wildcards. Pair the resulting value with `ESCAPE '\'` in the SQL — a
|
||||
/// single backslash, which under `standard_conforming_strings` (Postgres
|
||||
/// default) is one backslash in a single-quoted literal.
|
||||
///
|
||||
/// This is a search-correctness fix, not an injection fix: every term is
|
||||
/// already a bound parameter, so `%`/`_` can never break out of the string —
|
||||
/// they were just silently acting as wildcards (`50%` matching everything,
|
||||
/// `a_b` matching `axb`). Callers that build a substring pattern wrap the
|
||||
/// escaped term themselves, e.g. `format!("%{}%", escape_like(term))`.
|
||||
pub(crate) fn escape_like(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
for ch in s.chars() {
|
||||
if matches!(ch, '\\' | '%' | '_') {
|
||||
out.push('\\');
|
||||
}
|
||||
out.push(ch);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::escape_like;
|
||||
|
||||
#[test]
|
||||
fn escapes_wildcards_and_the_escape_char() {
|
||||
assert_eq!(escape_like("50%"), r"50\%");
|
||||
assert_eq!(escape_like("a_b"), r"a\_b");
|
||||
assert_eq!(escape_like(r"back\slash"), r"back\\slash");
|
||||
// A already-escaped-looking input is double-escaped so it stays literal.
|
||||
assert_eq!(escape_like(r"\%"), r"\\\%");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn leaves_ordinary_text_untouched() {
|
||||
assert_eq!(escape_like("naruto"), "naruto");
|
||||
assert_eq!(escape_like(""), "");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -78,13 +78,30 @@ pub enum EnqueueForPageOutcome {
|
||||
/// with `force=true`, running an UPDATE that flips the existing pending
|
||||
/// row's `force` flag to `true`; (3) reporting which path ran so the
|
||||
/// caller can audit accurately.
|
||||
/// Pool convenience wrapper for [`enqueue_for_page_conn`]. Use this from
|
||||
/// callers that don't need to share a transaction (chapter upload, the
|
||||
/// crawler). The admin force-reanalyze handler uses the `_conn` form so the
|
||||
/// enqueue and its `admin_audit` row commit together.
|
||||
pub async fn enqueue_for_page(
|
||||
pool: &PgPool,
|
||||
page_id: Uuid,
|
||||
force: bool,
|
||||
) -> AppResult<EnqueueForPageOutcome> {
|
||||
let mut conn = pool.acquire().await?;
|
||||
enqueue_for_page_conn(&mut conn, page_id, force).await
|
||||
}
|
||||
|
||||
/// Enqueue (or force-upgrade) an `analyze_page` job on a caller-supplied
|
||||
/// connection, so the admin handler can run it inside the same transaction as
|
||||
/// its audit insert. The body is a sequence of single statements, each
|
||||
/// reborrowing `&mut *conn`.
|
||||
pub async fn enqueue_for_page_conn(
|
||||
conn: &mut sqlx::PgConnection,
|
||||
page_id: Uuid,
|
||||
force: bool,
|
||||
) -> AppResult<EnqueueForPageOutcome> {
|
||||
use crate::crawler::jobs::EnqueueResult;
|
||||
match jobs::enqueue(pool, &JobPayload::AnalyzePage { page_id, force }).await? {
|
||||
match jobs::enqueue(&mut *conn, &JobPayload::AnalyzePage { page_id, force }).await? {
|
||||
EnqueueResult::Inserted(_) => Ok(EnqueueForPageOutcome::Inserted),
|
||||
EnqueueResult::Skipped if !force => Ok(EnqueueForPageOutcome::AlreadyEnqueued),
|
||||
EnqueueResult::Skipped => {
|
||||
@@ -106,7 +123,7 @@ pub async fn enqueue_for_page(
|
||||
RETURNING id, state",
|
||||
)
|
||||
.bind(page_id.to_string())
|
||||
.fetch_all(pool)
|
||||
.fetch_all(&mut *conn)
|
||||
.await?;
|
||||
if upgraded.is_empty() {
|
||||
// Race: between the skipped INSERT and the UPDATE the
|
||||
@@ -115,7 +132,9 @@ pub async fn enqueue_for_page(
|
||||
// would succeed. Retry once to close the race without
|
||||
// unbounded loops.
|
||||
return Ok(
|
||||
match jobs::enqueue(pool, &JobPayload::AnalyzePage { page_id, force }).await? {
|
||||
match jobs::enqueue(&mut *conn, &JobPayload::AnalyzePage { page_id, force })
|
||||
.await?
|
||||
{
|
||||
EnqueueResult::Inserted(_) => EnqueueForPageOutcome::Inserted,
|
||||
EnqueueResult::Skipped => EnqueueForPageOutcome::AlreadyEnqueued,
|
||||
},
|
||||
@@ -139,7 +158,7 @@ pub async fn enqueue_for_page(
|
||||
// the original's ack would clobber it.
|
||||
for (id, state) in &upgraded {
|
||||
if state == "running" {
|
||||
let _ = jobs::release_unowned(pool, *id).await;
|
||||
let _ = jobs::release_unowned(&mut *conn, *id).await;
|
||||
}
|
||||
}
|
||||
Ok(EnqueueForPageOutcome::UpgradedToForce)
|
||||
@@ -168,11 +187,14 @@ pub enum ReenqueueScope {
|
||||
/// skip-if-done net would no-op them). Pages with a pending/running
|
||||
/// `analyze_page` job are always skipped so repeated calls don't pile up
|
||||
/// duplicates. Returns the number of jobs enqueued.
|
||||
pub async fn enqueue_pages(
|
||||
pool: &PgPool,
|
||||
pub async fn enqueue_pages<'e, E>(
|
||||
executor: E,
|
||||
scope: ReenqueueScope,
|
||||
only_unanalyzed: bool,
|
||||
) -> AppResult<u64> {
|
||||
) -> AppResult<u64>
|
||||
where
|
||||
E: sqlx::PgExecutor<'e>,
|
||||
{
|
||||
// Scope predicate; the bound uuid (when present) is always $2.
|
||||
let scope_clause = match scope {
|
||||
ReenqueueScope::All => "",
|
||||
@@ -206,7 +228,7 @@ pub async fn enqueue_pages(
|
||||
ReenqueueScope::All => query,
|
||||
ReenqueueScope::Manga(id) | ReenqueueScope::Chapter(id) => query.bind(id),
|
||||
};
|
||||
Ok(query.execute(pool).await?.rows_affected())
|
||||
Ok(query.execute(executor).await?.rows_affected())
|
||||
}
|
||||
|
||||
/// Per-manga analysis coverage for the admin overview. Only mangas that
|
||||
@@ -219,6 +241,10 @@ pub async fn manga_coverage(
|
||||
limit: i64,
|
||||
offset: i64,
|
||||
) -> AppResult<(Vec<MangaCoverage>, i64)> {
|
||||
// LIKE-escape so `%`/`_` in the title filter match literally. No trigram
|
||||
// branch here, so binding the escaped term directly (rather than appending a
|
||||
// second param) is safe — $1 feeds only the ILIKE.
|
||||
let search = search.map(crate::repo::escape_like);
|
||||
let rows = sqlx::query_as::<_, MangaCoverage>(
|
||||
r#"
|
||||
SELECT m.id AS manga_id, m.title,
|
||||
@@ -228,13 +254,13 @@ pub async fn manga_coverage(
|
||||
JOIN chapters c ON c.manga_id = m.id
|
||||
JOIN pages p ON p.chapter_id = c.id
|
||||
LEFT JOIN page_analysis pa ON pa.page_id = p.id
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE '%' || $1 || '%')
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE '%' || $1 || '%' ESCAPE '\')
|
||||
GROUP BY m.id, m.title
|
||||
ORDER BY lower(m.title), m.id
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
)
|
||||
.bind(search)
|
||||
.bind(&search)
|
||||
.bind(limit)
|
||||
.bind(offset)
|
||||
.fetch_all(pool)
|
||||
@@ -247,12 +273,12 @@ pub async fn manga_coverage(
|
||||
FROM mangas m
|
||||
JOIN chapters c ON c.manga_id = m.id
|
||||
JOIN pages p ON p.chapter_id = c.id
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE '%' || $1 || '%')
|
||||
WHERE ($1::text IS NULL OR m.title ILIKE '%' || $1 || '%' ESCAPE '\')
|
||||
GROUP BY m.id
|
||||
) x
|
||||
"#,
|
||||
)
|
||||
.bind(search)
|
||||
.bind(&search)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
Ok((rows, total))
|
||||
@@ -355,7 +381,7 @@ pub async fn list_history(
|
||||
) -> AppResult<(Vec<AnalysisHistoryRow>, i64)> {
|
||||
let search_pat = filter
|
||||
.search
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
let items = sqlx::query_as::<_, AnalysisHistoryRow>(
|
||||
@@ -380,7 +406,7 @@ pub async fn list_history(
|
||||
WHERE pa.status <> 'pending'
|
||||
AND ($1::text IS NULL OR pa.status = $1)
|
||||
AND ($2::bool IS FALSE OR pa.is_nsfw)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 ESCAPE '\')
|
||||
ORDER BY pa.analyzed_at DESC NULLS LAST, pa.page_id
|
||||
LIMIT $4 OFFSET $5
|
||||
"#,
|
||||
@@ -403,7 +429,7 @@ pub async fn list_history(
|
||||
WHERE pa.status <> 'pending'
|
||||
AND ($1::text IS NULL OR pa.status = $1)
|
||||
AND ($2::bool IS FALSE OR pa.is_nsfw)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3)
|
||||
AND ($3::text IS NULL OR m.title ILIKE $3 ESCAPE '\')
|
||||
"#,
|
||||
)
|
||||
.bind(filter.status)
|
||||
|
||||
@@ -120,27 +120,11 @@ pub async fn list_for_page(
|
||||
Ok(rows.into_iter().map(|(t,)| t).collect())
|
||||
}
|
||||
|
||||
/// Escape a string for use as a LIKE pattern fragment: `%`, `_`, and
|
||||
/// `\` get a leading backslash so they're matched literally rather
|
||||
/// than as wildcards / escapes. The matching queries below pair this
|
||||
/// with `ESCAPE '\'` for explicitness — a single backslash, since the
|
||||
/// SQL lives in a raw string and Postgres treats `\\` in a single-
|
||||
/// quoted literal as one backslash under `standard_conforming_strings`.
|
||||
///
|
||||
/// The public API rejects `%`/`_`/`\` in `normalize_tag` before
|
||||
/// they reach this repo, so this is defence-in-depth — a future
|
||||
/// internal caller (worker, CLI) that bypasses the normalizer can't
|
||||
/// turn a prefix filter into a wildcard search by accident.
|
||||
fn escape_like_prefix(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
for ch in s.chars() {
|
||||
if matches!(ch, '\\' | '%' | '_') {
|
||||
out.push('\\');
|
||||
}
|
||||
out.push(ch);
|
||||
}
|
||||
out
|
||||
}
|
||||
// LIKE-escaping the autocomplete prefix is defence-in-depth: the public API
|
||||
// already rejects `%`/`_`/`\` in `normalize_tag` before they reach this repo, so
|
||||
// a stray wildcard can only arrive from a future internal caller (worker, CLI)
|
||||
// that bypasses the normalizer. Shared with the other search sites.
|
||||
use crate::repo::escape_like as escape_like_prefix;
|
||||
|
||||
/// Paged list of `user_id`'s tagged pages, with breadcrumb. When
|
||||
/// `tag_filter` is `Some(_)`, restrict to that exact tag (used by the
|
||||
@@ -247,6 +231,11 @@ pub async fn distinct_tags_for_user(
|
||||
/// storage keys (page-number ascending) so the row can render a
|
||||
/// thumbnail strip without a follow-up fetch.
|
||||
///
|
||||
/// When `text` is non-blank, results are further restricted to pages whose
|
||||
/// analysis `search_doc` matches the query (OCR text search), and both
|
||||
/// `match_count` and the sample thumbnails reflect that filtered set —
|
||||
/// i.e. `match_count` counts tagged pages whose OCR also matches `text`.
|
||||
///
|
||||
/// `order` is inlined via `format!()` — the enum value space is
|
||||
/// closed (`ASC` / `DESC`) so this is not a SQL-injection vector.
|
||||
pub async fn aggregate_chapters_for_tag(
|
||||
@@ -256,7 +245,30 @@ pub async fn aggregate_chapters_for_tag(
|
||||
order: Order,
|
||||
limit: i64,
|
||||
offset: i64,
|
||||
text: Option<&str>,
|
||||
) -> AppResult<(Vec<TaggedChapterAggregate>, i64)> {
|
||||
// OCR text filter: when present, additionally require the page's analysis
|
||||
// `search_doc` to match the query. Reuses the precomputed tsvector exactly
|
||||
// like `repo::page_analysis::page_search`. `None`/blank ⇒ tag-only.
|
||||
let text = text.map(str::trim).filter(|s| !s.is_empty());
|
||||
let (text_join, text_where) = match text {
|
||||
// `$5` in the main query, `$3` in the count query (see binds below).
|
||||
Some(_) => (
|
||||
"JOIN page_analysis pa ON pa.page_id = p.id",
|
||||
"AND pa.search_doc @@ plainto_tsquery('simple', {n})",
|
||||
),
|
||||
None => ("", ""),
|
||||
};
|
||||
// Same filter inside the correlated sample-thumbnail subquery (its page is
|
||||
// aliased `p`), so the thumbnails match what the text search matched. The
|
||||
// subquery lives in the main query, so it reuses the `$5` text bind.
|
||||
let (sample_join, sample_where) = match text {
|
||||
Some(_) => (
|
||||
"JOIN page_analysis pa2 ON pa2.page_id = p.id",
|
||||
"AND pa2.search_doc @@ plainto_tsquery('simple', $5)",
|
||||
),
|
||||
None => ("", ""),
|
||||
};
|
||||
let sql = format!(
|
||||
r#"
|
||||
SELECT
|
||||
@@ -274,9 +286,11 @@ pub async fn aggregate_chapters_for_tag(
|
||||
FROM pages p
|
||||
JOIN page_tags pt2 ON pt2.page_id = p.id
|
||||
JOIN tags t2 ON t2.id = pt2.tag_id
|
||||
{sample_join}
|
||||
WHERE p.chapter_id = ch.id
|
||||
AND pt2.user_id = $1
|
||||
AND lower(t2.name) = $2
|
||||
{sample_where}
|
||||
ORDER BY p.page_number ASC
|
||||
LIMIT 3
|
||||
) p2
|
||||
@@ -288,43 +302,60 @@ pub async fn aggregate_chapters_for_tag(
|
||||
JOIN pages p ON p.id = pt.page_id
|
||||
JOIN chapters ch ON ch.id = p.chapter_id
|
||||
JOIN mangas m ON m.id = ch.manga_id
|
||||
{text_join}
|
||||
WHERE pt.user_id = $1
|
||||
AND lower(t.name) = $2
|
||||
{text_where}
|
||||
GROUP BY ch.id, ch.manga_id, m.title, ch.number, ch.title
|
||||
ORDER BY match_count {dir}, ch.id
|
||||
LIMIT $3 OFFSET $4
|
||||
"#,
|
||||
dir = order.as_sql(),
|
||||
text_join = text_join,
|
||||
text_where = text_where.replace("{n}", "$5"),
|
||||
sample_join = sample_join,
|
||||
sample_where = sample_where,
|
||||
);
|
||||
let rows = sqlx::query_as::<_, TaggedChapterAggregate>(&sql)
|
||||
let mut q = sqlx::query_as::<_, TaggedChapterAggregate>(&sql)
|
||||
.bind(user_id)
|
||||
.bind(tag)
|
||||
.bind(limit)
|
||||
.bind(offset)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
.bind(offset);
|
||||
if let Some(text) = text {
|
||||
q = q.bind(text);
|
||||
}
|
||||
let rows = q.fetch_all(pool).await?;
|
||||
|
||||
let (total,): (i64,) = sqlx::query_as(
|
||||
let count_sql = format!(
|
||||
r#"
|
||||
SELECT count(*) FROM (
|
||||
SELECT 1
|
||||
FROM page_tags pt
|
||||
JOIN tags t ON t.id = pt.tag_id
|
||||
JOIN pages p ON p.id = pt.page_id
|
||||
{text_join}
|
||||
WHERE pt.user_id = $1 AND lower(t.name) = $2
|
||||
{text_where}
|
||||
GROUP BY p.chapter_id
|
||||
) c
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(tag)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
text_join = text_join,
|
||||
text_where = text_where.replace("{n}", "$3"),
|
||||
);
|
||||
let mut cq = sqlx::query_as::<_, (i64,)>(&count_sql).bind(user_id).bind(tag);
|
||||
if let Some(text) = text {
|
||||
cq = cq.bind(text);
|
||||
}
|
||||
let (total,) = cq.fetch_one(pool).await?;
|
||||
Ok((rows, total))
|
||||
}
|
||||
|
||||
/// Paged list of mangas containing pages tagged `tag` for `user_id`,
|
||||
/// ranked by `match_count` summed across all their chapters.
|
||||
///
|
||||
/// `text` behaves as in [`aggregate_chapters_for_tag`]: non-blank restricts to
|
||||
/// pages whose OCR `search_doc` matches, and both `match_count` and the sample
|
||||
/// thumbnails reflect that filtered set.
|
||||
pub async fn aggregate_mangas_for_tag(
|
||||
pool: &PgPool,
|
||||
user_id: Uuid,
|
||||
@@ -332,7 +363,26 @@ pub async fn aggregate_mangas_for_tag(
|
||||
order: Order,
|
||||
limit: i64,
|
||||
offset: i64,
|
||||
text: Option<&str>,
|
||||
) -> AppResult<(Vec<TaggedMangaAggregate>, i64)> {
|
||||
// OCR text filter — see `aggregate_chapters_for_tag` for the rationale.
|
||||
let text = text.map(str::trim).filter(|s| !s.is_empty());
|
||||
let (text_join, text_where) = match text {
|
||||
Some(_) => (
|
||||
"JOIN page_analysis pa ON pa.page_id = p.id",
|
||||
"AND pa.search_doc @@ plainto_tsquery('simple', {n})",
|
||||
),
|
||||
None => ("", ""),
|
||||
};
|
||||
// Same filter inside the sample-thumbnail subquery (page aliased `p`),
|
||||
// reusing the main query's `$5` text bind.
|
||||
let (sample_join, sample_where) = match text {
|
||||
Some(_) => (
|
||||
"JOIN page_analysis pa2 ON pa2.page_id = p.id",
|
||||
"AND pa2.search_doc @@ plainto_tsquery('simple', $5)",
|
||||
),
|
||||
None => ("", ""),
|
||||
};
|
||||
let sql = format!(
|
||||
r#"
|
||||
SELECT
|
||||
@@ -349,9 +399,11 @@ pub async fn aggregate_mangas_for_tag(
|
||||
JOIN chapters ch2 ON ch2.id = p.chapter_id
|
||||
JOIN page_tags pt2 ON pt2.page_id = p.id
|
||||
JOIN tags t2 ON t2.id = pt2.tag_id
|
||||
{sample_join}
|
||||
WHERE ch2.manga_id = m.id
|
||||
AND pt2.user_id = $1
|
||||
AND lower(t2.name) = $2
|
||||
{sample_where}
|
||||
ORDER BY p.page_number ASC
|
||||
LIMIT 3
|
||||
) p2
|
||||
@@ -363,23 +415,31 @@ pub async fn aggregate_mangas_for_tag(
|
||||
JOIN pages p ON p.id = pt.page_id
|
||||
JOIN chapters ch ON ch.id = p.chapter_id
|
||||
JOIN mangas m ON m.id = ch.manga_id
|
||||
{text_join}
|
||||
WHERE pt.user_id = $1
|
||||
AND lower(t.name) = $2
|
||||
{text_where}
|
||||
GROUP BY m.id, m.title, m.cover_image_path
|
||||
ORDER BY match_count {dir}, m.id
|
||||
LIMIT $3 OFFSET $4
|
||||
"#,
|
||||
dir = order.as_sql(),
|
||||
text_join = text_join,
|
||||
text_where = text_where.replace("{n}", "$5"),
|
||||
sample_join = sample_join,
|
||||
sample_where = sample_where,
|
||||
);
|
||||
let rows = sqlx::query_as::<_, TaggedMangaAggregate>(&sql)
|
||||
let mut q = sqlx::query_as::<_, TaggedMangaAggregate>(&sql)
|
||||
.bind(user_id)
|
||||
.bind(tag)
|
||||
.bind(limit)
|
||||
.bind(offset)
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
.bind(offset);
|
||||
if let Some(text) = text {
|
||||
q = q.bind(text);
|
||||
}
|
||||
let rows = q.fetch_all(pool).await?;
|
||||
|
||||
let (total,): (i64,) = sqlx::query_as(
|
||||
let count_sql = format!(
|
||||
r#"
|
||||
SELECT count(*) FROM (
|
||||
SELECT 1
|
||||
@@ -387,15 +447,20 @@ pub async fn aggregate_mangas_for_tag(
|
||||
JOIN tags t ON t.id = pt.tag_id
|
||||
JOIN pages p ON p.id = pt.page_id
|
||||
JOIN chapters ch ON ch.id = p.chapter_id
|
||||
{text_join}
|
||||
WHERE pt.user_id = $1 AND lower(t.name) = $2
|
||||
{text_where}
|
||||
GROUP BY ch.manga_id
|
||||
) m
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(tag)
|
||||
.fetch_one(pool)
|
||||
.await?;
|
||||
text_join = text_join,
|
||||
text_where = text_where.replace("{n}", "$3"),
|
||||
);
|
||||
let mut cq = sqlx::query_as::<_, (i64,)>(&count_sql).bind(user_id).bind(tag);
|
||||
if let Some(text) = text {
|
||||
cq = cq.bind(text);
|
||||
}
|
||||
let (total,) = cq.fetch_one(pool).await?;
|
||||
Ok((rows, total))
|
||||
}
|
||||
|
||||
|
||||
66
backend/src/repo/reaction.rs
Normal file
66
backend/src/repo/reaction.rs
Normal file
@@ -0,0 +1,66 @@
|
||||
//! Per-user manga reaction (like/dislike) persistence.
|
||||
|
||||
use sqlx::PgPool;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::domain::reaction::Reaction;
|
||||
use crate::error::{AppError, AppResult};
|
||||
|
||||
/// Insert-or-overwrite the user's reaction on this manga (like ↔ dislike).
|
||||
/// A foreign-key violation (unknown manga) maps to `NotFound` so the API
|
||||
/// returns 404 rather than 500 — mirrors `read_progress::upsert`.
|
||||
pub async fn upsert(
|
||||
pool: &PgPool,
|
||||
user_id: Uuid,
|
||||
manga_id: Uuid,
|
||||
reaction: Reaction,
|
||||
) -> AppResult<()> {
|
||||
sqlx::query(
|
||||
r#"
|
||||
INSERT INTO manga_reactions (user_id, manga_id, reaction, created_at)
|
||||
VALUES ($1, $2, $3, now())
|
||||
ON CONFLICT (user_id, manga_id) DO UPDATE
|
||||
SET reaction = EXCLUDED.reaction,
|
||||
created_at = now()
|
||||
"#,
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(manga_id)
|
||||
.bind(reaction.as_str())
|
||||
.execute(pool)
|
||||
.await
|
||||
.map_err(|e| match e {
|
||||
sqlx::Error::Database(ref db_err) if db_err.is_foreign_key_violation() => {
|
||||
AppError::NotFound
|
||||
}
|
||||
other => AppError::Database(other),
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Remove the user's reaction on this manga. Idempotent — clearing a
|
||||
/// non-existent reaction is a no-op.
|
||||
pub async fn clear(pool: &PgPool, user_id: Uuid, manga_id: Uuid) -> AppResult<()> {
|
||||
sqlx::query("DELETE FROM manga_reactions WHERE user_id = $1 AND manga_id = $2")
|
||||
.bind(user_id)
|
||||
.bind(manga_id)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The user's current reaction on this manga, or `None` if unset.
|
||||
pub async fn get(
|
||||
pool: &PgPool,
|
||||
user_id: Uuid,
|
||||
manga_id: Uuid,
|
||||
) -> AppResult<Option<Reaction>> {
|
||||
let row: Option<(String,)> = sqlx::query_as(
|
||||
"SELECT reaction FROM manga_reactions WHERE user_id = $1 AND manga_id = $2",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(manga_id)
|
||||
.fetch_optional(pool)
|
||||
.await?;
|
||||
Ok(row.and_then(|(s,)| Reaction::parse(&s)))
|
||||
}
|
||||
@@ -80,7 +80,17 @@ pub async fn get_for_manga(
|
||||
rp.chapter_id,
|
||||
c.number AS chapter_number,
|
||||
rp.page,
|
||||
rp.updated_at
|
||||
rp.updated_at,
|
||||
-- Distinct chapter numbers past the reader's last-read chapter
|
||||
-- (see list_for_user for the non-unique-number rationale). 0
|
||||
-- when the last-read chapter is unknown.
|
||||
(
|
||||
SELECT count(DISTINCT c2.number)
|
||||
FROM chapters c2
|
||||
WHERE c2.manga_id = rp.manga_id
|
||||
AND c.number IS NOT NULL
|
||||
AND c2.number > c.number
|
||||
) AS new_chapters_count
|
||||
FROM read_progress rp
|
||||
LEFT JOIN chapters c ON c.id = rp.chapter_id
|
||||
WHERE rp.user_id = $1 AND rp.manga_id = $2
|
||||
@@ -131,8 +141,23 @@ pub async fn list_for_user(
|
||||
m.cover_image_path AS manga_cover_image_path,
|
||||
rp.chapter_id,
|
||||
c.number AS chapter_number,
|
||||
c.page_count AS chapter_page_count,
|
||||
rp.page,
|
||||
rp.updated_at
|
||||
rp.updated_at,
|
||||
-- Personal "new since last read": how many distinct chapter
|
||||
-- numbers sit past the reader's last-read chapter.
|
||||
-- COUNT(DISTINCT number) — not COUNT(*) — because
|
||||
-- (manga_id, number) is non-unique (scanlations share a
|
||||
-- number, migration 0013), so raw rows would over-count. When
|
||||
-- `c.number` is NULL (manga-level progress or a deleted
|
||||
-- chapter) the predicate matches no rows, yielding 0.
|
||||
(
|
||||
SELECT count(DISTINCT c2.number)
|
||||
FROM chapters c2
|
||||
WHERE c2.manga_id = rp.manga_id
|
||||
AND c.number IS NOT NULL
|
||||
AND c2.number > c.number
|
||||
) AS new_chapters_count
|
||||
FROM read_progress rp
|
||||
JOIN mangas m ON m.id = rp.manga_id
|
||||
LEFT JOIN chapters c ON c.id = rp.chapter_id
|
||||
|
||||
@@ -65,3 +65,15 @@ pub async fn delete_by_token_hash(pool: &PgPool, token_hash: &[u8]) -> AppResult
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Delete every session whose `expires_at` has passed, returning the number
|
||||
/// reaped. `find_active` already refuses expired sessions, so this only
|
||||
/// reclaims storage — running it on any schedule (or not at all) is safe.
|
||||
/// Backed by `sessions_expires_idx` (0002). Called by the periodic reaper in
|
||||
/// `app::build`.
|
||||
pub async fn delete_expired(pool: &PgPool) -> AppResult<u64> {
|
||||
let result = sqlx::query("DELETE FROM sessions WHERE expires_at <= now()")
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(result.rows_affected())
|
||||
}
|
||||
|
||||
@@ -124,7 +124,7 @@ pub async fn list(
|
||||
SELECT id, name, created_at
|
||||
FROM tags
|
||||
WHERE $1::text IS NULL
|
||||
OR name ILIKE '%' || $1 || '%'
|
||||
OR name ILIKE '%' || $3 || '%' ESCAPE '\'
|
||||
OR name % $1
|
||||
ORDER BY CASE WHEN $1::text IS NULL THEN 0 ELSE similarity(name, $1) END DESC,
|
||||
lower(name) ASC
|
||||
@@ -133,6 +133,9 @@ pub async fn list(
|
||||
)
|
||||
.bind(search)
|
||||
.bind(limit)
|
||||
// $3: LIKE-escaped term for the ILIKE branch so `%`/`_` match literally; the
|
||||
// trigram/similarity branches keep the raw $1.
|
||||
.bind(search.map(crate::repo::escape_like))
|
||||
.fetch_all(pool)
|
||||
.await?;
|
||||
Ok(rows)
|
||||
|
||||
@@ -110,7 +110,7 @@ pub async fn list_for_user(
|
||||
});
|
||||
}
|
||||
// Newest first; trim to limit after the merge.
|
||||
entries.sort_by(|a, b| b.created_at().cmp(&a.created_at()));
|
||||
entries.sort_by_key(|b| std::cmp::Reverse(b.created_at()));
|
||||
entries.truncate(limit as usize);
|
||||
|
||||
let (manga_total, chapter_total): (i64, i64) = sqlx::query_as(
|
||||
|
||||
@@ -106,14 +106,14 @@ pub async fn list_with_total(
|
||||
let pat = q
|
||||
.search
|
||||
.as_ref()
|
||||
.map(|s| format!("%{}%", s.trim()))
|
||||
.map(|s| format!("%{}%", crate::repo::escape_like(s.trim())))
|
||||
.filter(|p| p.len() > 2);
|
||||
|
||||
let items = sqlx::query_as::<_, User>(
|
||||
r#"
|
||||
SELECT id, username, password_hash, created_at, is_admin
|
||||
FROM users
|
||||
WHERE ($1::text IS NULL OR username ILIKE $1)
|
||||
WHERE ($1::text IS NULL OR username ILIKE $1 ESCAPE '\')
|
||||
ORDER BY username
|
||||
LIMIT $2 OFFSET $3
|
||||
"#,
|
||||
@@ -125,7 +125,7 @@ pub async fn list_with_total(
|
||||
.await?;
|
||||
|
||||
let total: i64 = sqlx::query_scalar(
|
||||
"SELECT COUNT(*) FROM users WHERE ($1::text IS NULL OR username ILIKE $1)",
|
||||
"SELECT COUNT(*) FROM users WHERE ($1::text IS NULL OR username ILIKE $1 ESCAPE '\\')",
|
||||
)
|
||||
.bind(&pat)
|
||||
.fetch_one(pool)
|
||||
@@ -157,8 +157,10 @@ pub async fn set_is_admin_unchecked(pool: &PgPool, id: Uuid, value: bool) -> App
|
||||
/// - If a row already exists: flip `is_admin` to true if needed; **never**
|
||||
/// touch the existing `password_hash`. Lets the operator rotate the
|
||||
/// admin password through the UI without env-var conflict.
|
||||
///
|
||||
/// Wrapped in a transaction so a concurrent `register` for the same
|
||||
/// username can't slip an INSERT between the SELECT and UPDATE/INSERT.
|
||||
///
|
||||
/// Set `is_admin` on a user with full safety checks: rejects self-demote,
|
||||
/// rejects demoting the only remaining admin (under `ADMIN_INVARIANT_LOCK_KEY`
|
||||
/// to close the parallel-demote race), and writes an `admin_audit` row
|
||||
@@ -379,7 +381,7 @@ pub async fn bootstrap_admin(
|
||||
.await?;
|
||||
}
|
||||
None => {
|
||||
let hash = crate::auth::password::hash_password(password)?;
|
||||
let hash = crate::auth::password::hash_password_async(password.to_string()).await?;
|
||||
sqlx::query("INSERT INTO users (username, password_hash, is_admin) VALUES ($1, $2, true)")
|
||||
.bind(username)
|
||||
.bind(&hash)
|
||||
|
||||
@@ -25,7 +25,7 @@ use serde::{Deserialize, Serialize};
|
||||
use crate::analysis::prompt::{
|
||||
GROUNDING_PROMPT_DEFAULT, OCR_PROMPT_DEFAULT, SYSTEM_PROMPT_DEFAULT,
|
||||
};
|
||||
use crate::config::{AnalysisConfig, CrawlerConfig, ResponseFormat};
|
||||
use crate::config::{AnalysisBackend, AnalysisConfig, CrawlerConfig, ResponseFormat};
|
||||
use crate::crawler::safety::DownloadAllowlist;
|
||||
|
||||
/// `app_settings.key` for the crawler group.
|
||||
@@ -33,6 +33,28 @@ pub const KEY_CRAWLER: &str = "crawler";
|
||||
/// `app_settings.key` for the analysis group.
|
||||
pub const KEY_ANALYSIS: &str = "analysis";
|
||||
|
||||
// Upper bounds on numeric settings. These are sanity caps, not tuned limits —
|
||||
// they keep a fat-fingered (or CSRF-injected) value from spawning thousands of
|
||||
// workers, demanding gigabyte buffers, or sending out-of-range sampling
|
||||
// params. Generous enough that no realistic deployment hits them.
|
||||
const MAX_WORKERS: u64 = 64;
|
||||
const MAX_CHAPTER_WORKERS: u64 = 64;
|
||||
const MAX_ANALYSIS_MAX_TOKENS: u32 = 1_000_000;
|
||||
const MAX_ANALYSIS_SLICES: u64 = 1024;
|
||||
/// 1 GiB — far above any real page, well below "exhaust the host".
|
||||
const MAX_ANALYSIS_IMAGE_BYTES: u64 = 1024 * 1024 * 1024;
|
||||
/// OpenAI-compatible sampling ranges (the analysis endpoint speaks that API).
|
||||
const MAX_TEMPERATURE: f64 = 2.0;
|
||||
const FREQUENCY_PENALTY_MIN: f64 = -2.0;
|
||||
const FREQUENCY_PENALTY_MAX: f64 = 2.0;
|
||||
/// Max manga-detail fetches per metadata pass. `0` stays special-cased as
|
||||
/// "unlimited"; this only bounds an explicit positive value.
|
||||
const MAX_MANGA_LIMIT: u64 = 1_000_000;
|
||||
/// Upper bound (seconds) shared by every timeout knob — 24h. A timeout
|
||||
/// longer than a day is almost certainly a fat-fingered value (e.g. ms
|
||||
/// mistaken for s) and would wedge a worker for far too long.
|
||||
const MAX_TIMEOUT_SECS: u64 = 86_400;
|
||||
|
||||
/// One field-level validation failure, surfaced to the UI per input.
|
||||
#[derive(Debug, Clone, Serialize, PartialEq, Eq)]
|
||||
pub struct FieldError {
|
||||
@@ -151,6 +173,8 @@ impl CrawlerSettings {
|
||||
};
|
||||
if self.chapter_workers < 1 {
|
||||
errs.push("chapter_workers", "must be at least 1");
|
||||
} else if self.chapter_workers > MAX_CHAPTER_WORKERS {
|
||||
errs.push("chapter_workers", format!("must be at most {MAX_CHAPTER_WORKERS}"));
|
||||
}
|
||||
if let Some(url) = self.start_url.as_deref().map(str::trim).filter(|s| !s.is_empty()) {
|
||||
// SSRF defence: Url::parse alone admits http://169.254.169.254
|
||||
@@ -164,6 +188,16 @@ impl CrawlerSettings {
|
||||
}
|
||||
if self.job_timeout_secs < 1 {
|
||||
errs.push("job_timeout_secs", "must be at least 1 second");
|
||||
} else if self.job_timeout_secs > MAX_TIMEOUT_SECS {
|
||||
errs.push("job_timeout_secs", format!("must be at most {MAX_TIMEOUT_SECS} seconds"));
|
||||
}
|
||||
if self.idle_timeout_secs > MAX_TIMEOUT_SECS {
|
||||
errs.push("idle_timeout_secs", format!("must be at most {MAX_TIMEOUT_SECS} seconds"));
|
||||
}
|
||||
// 0 is intentionally "unlimited"; only an explicit positive value is
|
||||
// capped.
|
||||
if self.manga_limit > MAX_MANGA_LIMIT {
|
||||
errs.push("manga_limit", format!("must be at most {MAX_MANGA_LIMIT}"));
|
||||
}
|
||||
|
||||
if !errs.is_empty() {
|
||||
@@ -220,6 +254,9 @@ impl CrawlerSettings {
|
||||
.map(str::to_string),
|
||||
download_allowlist,
|
||||
max_image_bytes: self.max_image_bytes as usize,
|
||||
// Env-only safety cap, not a runtime-editable setting — preserve
|
||||
// it from the base so a settings reload keeps the boot value.
|
||||
max_images_per_chapter: base.max_images_per_chapter,
|
||||
manga_limit: self.manga_limit as usize,
|
||||
job_timeout: Duration::from_secs(self.job_timeout_secs.max(1)),
|
||||
metadata_max_consecutive_failures: self.metadata_max_consecutive_failures,
|
||||
@@ -232,6 +269,7 @@ impl CrawlerSettings {
|
||||
tor_control_cookie_path: base.tor_control_cookie_path.clone(),
|
||||
tor_recircuit_max_attempts: base.tor_recircuit_max_attempts,
|
||||
browser: base.browser.clone(),
|
||||
ssrf_intercept: base.ssrf_intercept,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -354,7 +392,18 @@ impl AnalysisSettings {
|
||||
|
||||
if self.workers < 1 {
|
||||
errs.push("workers", "must be at least 1");
|
||||
} else if self.workers > MAX_WORKERS {
|
||||
errs.push("workers", format!("must be at most {MAX_WORKERS}"));
|
||||
}
|
||||
// The endpoint/model are vision-only knobs — the OCR backend never
|
||||
// dials a URL or sends a model id. While vision is dormant
|
||||
// (`base.backend == Ocr`), skip the live-worker SSRF gate and the
|
||||
// model-required check so an OCR operator can enable the worker
|
||||
// without a vision endpoint/model (the OCR settings UI doesn't even
|
||||
// expose them). The basic malformed-URL sanity check below stays
|
||||
// unconditional. Re-enabling vision restores the full gate via the
|
||||
// `base.backend == Vision` predicate.
|
||||
let vision_active = base.backend == AnalysisBackend::Vision;
|
||||
let trimmed_endpoint = self.endpoint.trim();
|
||||
if trimmed_endpoint.is_empty() {
|
||||
errs.push("endpoint", "must be a valid absolute URL");
|
||||
@@ -362,7 +411,7 @@ impl AnalysisSettings {
|
||||
// Always reject obviously-malformed URLs (matches prior behaviour).
|
||||
let _ = e;
|
||||
errs.push("endpoint", "must be a valid absolute URL");
|
||||
} else if self.enabled {
|
||||
} else if self.enabled && vision_active {
|
||||
// SSRF + API-key exfiltration defence — only enforced when the
|
||||
// worker is actually live. Worker attaches an env-managed bearer
|
||||
// token to every call; without this check, an admin (or one
|
||||
@@ -379,23 +428,40 @@ impl AnalysisSettings {
|
||||
errs.push("endpoint", url_safety_message(&e));
|
||||
}
|
||||
}
|
||||
if self.enabled && self.model.trim().is_empty() {
|
||||
if self.enabled && vision_active && self.model.trim().is_empty() {
|
||||
errs.push("model", "required when analysis is enabled");
|
||||
}
|
||||
if self.max_tokens < 1 {
|
||||
errs.push("max_tokens", "must be at least 1");
|
||||
} else if self.max_tokens > MAX_ANALYSIS_MAX_TOKENS {
|
||||
errs.push("max_tokens", format!("must be at most {MAX_ANALYSIS_MAX_TOKENS}"));
|
||||
}
|
||||
if self.request_timeout_secs < 1 {
|
||||
errs.push("request_timeout_secs", "must be at least 1 second");
|
||||
} else if self.request_timeout_secs > MAX_TIMEOUT_SECS {
|
||||
errs.push(
|
||||
"request_timeout_secs",
|
||||
format!("must be at most {MAX_TIMEOUT_SECS} seconds"),
|
||||
);
|
||||
}
|
||||
if self.job_timeout_secs < 1 {
|
||||
errs.push("job_timeout_secs", "must be at least 1 second");
|
||||
} else if self.job_timeout_secs > MAX_TIMEOUT_SECS {
|
||||
errs.push(
|
||||
"job_timeout_secs",
|
||||
format!("must be at most {MAX_TIMEOUT_SECS} seconds"),
|
||||
);
|
||||
}
|
||||
if self.max_pixels < 1 {
|
||||
errs.push("max_pixels", "must be at least 1");
|
||||
}
|
||||
if self.max_image_bytes < 1 {
|
||||
errs.push("max_image_bytes", "must be greater than 0");
|
||||
} else if self.max_image_bytes > MAX_ANALYSIS_IMAGE_BYTES {
|
||||
errs.push(
|
||||
"max_image_bytes",
|
||||
format!("must be at most {MAX_ANALYSIS_IMAGE_BYTES} (1 GiB)"),
|
||||
);
|
||||
}
|
||||
if !(0.0..=0.9).contains(&self.slice_overlap) {
|
||||
errs.push("slice_overlap", "must be between 0.0 and 0.9");
|
||||
@@ -408,9 +474,17 @@ impl AnalysisSettings {
|
||||
}
|
||||
if self.max_slices < 1 {
|
||||
errs.push("max_slices", "must be at least 1");
|
||||
} else if self.max_slices > MAX_ANALYSIS_SLICES {
|
||||
errs.push("max_slices", format!("must be at most {MAX_ANALYSIS_SLICES}"));
|
||||
}
|
||||
if self.temperature < 0.0 {
|
||||
errs.push("temperature", "must be 0 or greater");
|
||||
if !(0.0..=MAX_TEMPERATURE).contains(&self.temperature) {
|
||||
errs.push("temperature", format!("must be between 0 and {MAX_TEMPERATURE}"));
|
||||
}
|
||||
if !(FREQUENCY_PENALTY_MIN..=FREQUENCY_PENALTY_MAX).contains(&self.frequency_penalty) {
|
||||
errs.push(
|
||||
"frequency_penalty",
|
||||
format!("must be between {FREQUENCY_PENALTY_MIN} and {FREQUENCY_PENALTY_MAX}"),
|
||||
);
|
||||
}
|
||||
let response_format = match ResponseFormat::parse_strict(&self.response_format) {
|
||||
Some(rf) => rf,
|
||||
@@ -459,6 +533,13 @@ impl AnalysisSettings {
|
||||
api_key: base.api_key.clone(),
|
||||
// Env-only readiness probe URL preserved from the base.
|
||||
vision_health_url: base.vision_health_url.clone(),
|
||||
// Deploy-time engine selection + ocrs model paths: env-only, not
|
||||
// admin-tunable, so carry them through from the base unchanged.
|
||||
backend: base.backend,
|
||||
ocr_detection_model: base.ocr_detection_model.clone(),
|
||||
ocr_recognition_model: base.ocr_recognition_model.clone(),
|
||||
// Env-only decompression-bomb decode cap, carried from the base.
|
||||
ocr_max_decode_pixels: base.ocr_max_decode_pixels,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -490,11 +571,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn crawler_round_trips_through_dto() {
|
||||
let mut base = CrawlerConfig::default();
|
||||
base.start_url = Some("https://example.com/".to_string());
|
||||
base.tz = Tz::Europe__Berlin;
|
||||
base.chapter_workers = 3;
|
||||
base.cookie_domain = Some("example.com".to_string());
|
||||
let base = CrawlerConfig {
|
||||
start_url: Some("https://example.com/".to_string()),
|
||||
tz: Tz::Europe__Berlin,
|
||||
chapter_workers: 3,
|
||||
cookie_domain: Some("example.com".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
let dto = CrawlerSettings::from_config(&base);
|
||||
let back = dto.to_config(&base).expect("valid");
|
||||
assert_eq!(back.tz, Tz::Europe__Berlin);
|
||||
@@ -520,10 +603,12 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn crawler_overlay_preserves_env_only_fields() {
|
||||
let mut base = CrawlerConfig::default();
|
||||
base.proxy = Some("socks5://127.0.0.1:9050".to_string());
|
||||
base.tor_control_password = Some("secret".to_string());
|
||||
base.phpsessid = Some("abc123".to_string());
|
||||
let base = CrawlerConfig {
|
||||
proxy: Some("socks5://127.0.0.1:9050".to_string()),
|
||||
tor_control_password: Some("secret".to_string()),
|
||||
phpsessid: Some("abc123".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
// A DTO that knows nothing about the env-only fields.
|
||||
let dto = CrawlerSettings {
|
||||
rate_ms: 2000,
|
||||
@@ -627,8 +712,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn analysis_captures_env_prompt_override() {
|
||||
let mut base = AnalysisConfig::default();
|
||||
base.system_prompt = "custom env prompt".to_string();
|
||||
let base = AnalysisConfig {
|
||||
system_prompt: "custom env prompt".to_string(),
|
||||
..Default::default()
|
||||
};
|
||||
let dto = AnalysisSettings::from_config(&base);
|
||||
assert_eq!(dto.system_prompt.as_deref(), Some("custom env prompt"));
|
||||
}
|
||||
@@ -652,8 +739,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn analysis_overlay_preserves_api_key() {
|
||||
let mut base = AnalysisConfig::default();
|
||||
base.api_key = Some("sk-secret".to_string());
|
||||
let base = AnalysisConfig {
|
||||
api_key: Some("sk-secret".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
let dto = AnalysisSettings::from_config(&base);
|
||||
assert_eq!(dto.to_config(&base).unwrap().api_key.as_deref(), Some("sk-secret"));
|
||||
}
|
||||
@@ -690,13 +779,115 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_rejects_over_upper_bounds() {
|
||||
// Sanity caps: an absurdly large worker count / buffer / token budget
|
||||
// and out-of-range sampling params are all refused so a fat-fingered
|
||||
// or CSRF-injected value can't exhaust the host or break the upstream
|
||||
// API contract.
|
||||
let base = AnalysisConfig::default();
|
||||
let dto = AnalysisSettings {
|
||||
workers: MAX_WORKERS + 1,
|
||||
max_tokens: MAX_ANALYSIS_MAX_TOKENS + 1,
|
||||
max_slices: MAX_ANALYSIS_SLICES + 1,
|
||||
max_image_bytes: MAX_ANALYSIS_IMAGE_BYTES + 1,
|
||||
temperature: MAX_TEMPERATURE + 0.5,
|
||||
frequency_penalty: FREQUENCY_PENALTY_MAX + 0.5,
|
||||
request_timeout_secs: MAX_TIMEOUT_SECS + 1,
|
||||
job_timeout_secs: MAX_TIMEOUT_SECS + 1,
|
||||
..AnalysisSettings::from_config(&base)
|
||||
};
|
||||
let errs = dto.to_config(&base).unwrap_err();
|
||||
let fields: Vec<_> = errs.errors.iter().map(|e| e.field.as_str()).collect();
|
||||
for f in [
|
||||
"workers",
|
||||
"max_tokens",
|
||||
"max_slices",
|
||||
"max_image_bytes",
|
||||
"temperature",
|
||||
"frequency_penalty",
|
||||
"request_timeout_secs",
|
||||
"job_timeout_secs",
|
||||
] {
|
||||
assert!(fields.contains(&f), "missing upper-bound error for {f}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn crawler_rejects_over_upper_bounds() {
|
||||
// manga_limit ceiling + timeout caps. manga_limit=0 stays "unlimited"
|
||||
// and is asserted valid by the round-trip tests above.
|
||||
let base = CrawlerConfig::default();
|
||||
let dto = CrawlerSettings {
|
||||
manga_limit: MAX_MANGA_LIMIT + 1,
|
||||
job_timeout_secs: MAX_TIMEOUT_SECS + 1,
|
||||
idle_timeout_secs: MAX_TIMEOUT_SECS + 1,
|
||||
..CrawlerSettings::from_config(&base)
|
||||
};
|
||||
let errs = dto.to_config(&base).unwrap_err();
|
||||
let fields: Vec<_> = errs.errors.iter().map(|e| e.field.as_str()).collect();
|
||||
for f in ["manga_limit", "job_timeout_secs", "idle_timeout_secs"] {
|
||||
assert!(fields.contains(&f), "missing upper-bound error for {f}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn crawler_allows_unlimited_manga_limit() {
|
||||
// 0 means "no cap" and must stay valid.
|
||||
let base = CrawlerConfig::default();
|
||||
let dto = CrawlerSettings {
|
||||
manga_limit: 0,
|
||||
..CrawlerSettings::from_config(&base)
|
||||
};
|
||||
assert!(dto.to_config(&base).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_rejects_negative_frequency_penalty_below_range() {
|
||||
let base = AnalysisConfig::default();
|
||||
let dto = AnalysisSettings {
|
||||
frequency_penalty: FREQUENCY_PENALTY_MIN - 0.5,
|
||||
..AnalysisSettings::from_config(&base)
|
||||
};
|
||||
let errs = dto.to_config(&base).unwrap_err();
|
||||
let fields: Vec<_> = errs.errors.iter().map(|e| e.field.as_str()).collect();
|
||||
assert!(fields.contains(&"frequency_penalty"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_accepts_in_range_sampling_params() {
|
||||
// A normal config with mid-range sampling values stays valid.
|
||||
let base = AnalysisConfig::default();
|
||||
let dto = AnalysisSettings {
|
||||
temperature: 0.7,
|
||||
frequency_penalty: 0.3,
|
||||
..AnalysisSettings::from_config(&base)
|
||||
};
|
||||
assert!(dto.to_config(&base).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn crawler_rejects_over_chapter_worker_cap() {
|
||||
let base = CrawlerConfig::default();
|
||||
let dto = CrawlerSettings {
|
||||
chapter_workers: MAX_CHAPTER_WORKERS + 1,
|
||||
..CrawlerSettings::from_config(&base)
|
||||
};
|
||||
let errs = dto.to_config(&base).unwrap_err();
|
||||
let fields: Vec<_> = errs.errors.iter().map(|e| e.field.as_str()).collect();
|
||||
assert!(fields.contains(&"chapter_workers"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_endpoint_rejects_ip_literal_attacks_when_enabled() {
|
||||
// The vision worker bearer-attaches an env secret to every call;
|
||||
// a hostile/CSRF-able admin must NOT be able to point endpoint at
|
||||
// cloud metadata, loopback services, or RFC1918 hosts — when the
|
||||
// worker is enabled (toggling enabled=true later re-runs this gate).
|
||||
let base = AnalysisConfig::default();
|
||||
let base = AnalysisConfig {
|
||||
backend: AnalysisBackend::Vision,
|
||||
..Default::default()
|
||||
};
|
||||
for url in [
|
||||
"http://169.254.169.254/v1/chat/completions",
|
||||
"http://127.0.0.1:5432/",
|
||||
@@ -720,7 +911,10 @@ mod tests {
|
||||
// The documented default — docker DNS name resolving to a private IP
|
||||
// at runtime — must still validate, because the bearer recipient
|
||||
// identity is the operator-chosen hostname, not the underlying IP.
|
||||
let base = AnalysisConfig::default();
|
||||
let base = AnalysisConfig {
|
||||
backend: AnalysisBackend::Vision,
|
||||
..Default::default()
|
||||
};
|
||||
for url in [
|
||||
"http://mangalord-vision:8000/v1/chat/completions",
|
||||
"https://api.openai.com/v1/chat/completions",
|
||||
@@ -751,7 +945,12 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn analysis_requires_model_only_when_enabled() {
|
||||
let base = AnalysisConfig::default();
|
||||
// Vision base: the model id is required only when the vision worker
|
||||
// is actually live (see the OCR carve-out below).
|
||||
let base = AnalysisConfig {
|
||||
backend: AnalysisBackend::Vision,
|
||||
..Default::default()
|
||||
};
|
||||
let disabled = AnalysisSettings {
|
||||
enabled: false,
|
||||
model: "".to_string(),
|
||||
@@ -767,6 +966,28 @@ mod tests {
|
||||
assert!(errs.errors.iter().any(|e| e.field == "model"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn analysis_ocr_backend_enable_skips_vision_endpoint_and_model_validation() {
|
||||
// With the OCR backend active the worker never dials the vision
|
||||
// endpoint nor sends a model id, so enabling analysis must NOT
|
||||
// validate those vision-only fields. The default base carries the
|
||||
// dev-localhost endpoint (which the SSRF gate would otherwise reject)
|
||||
// and an empty model — under OCR, enabling is still valid. Without
|
||||
// this carve-out an OCR operator can't turn the worker on, since the
|
||||
// OCR settings UI doesn't expose endpoint/model to fix.
|
||||
let base = AnalysisConfig::default(); // backend = Ocr
|
||||
let dto = AnalysisSettings {
|
||||
enabled: true,
|
||||
endpoint: "http://localhost:8000/v1/chat/completions".to_string(),
|
||||
model: "".to_string(),
|
||||
..AnalysisSettings::from_config(&base)
|
||||
};
|
||||
assert!(
|
||||
dto.to_config(&base).is_ok(),
|
||||
"OCR-enabled config must not fail on vision endpoint/model"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dtos_serialize_to_json_and_back() {
|
||||
let c = CrawlerSettings::default();
|
||||
|
||||
@@ -95,6 +95,11 @@ impl Storage for LocalStorage {
|
||||
match fs::read(&path).await {
|
||||
Ok(b) => Ok(b),
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => Err(StorageError::NotFound),
|
||||
// A key resolving to a directory isn't a stored blob; `fs::read`
|
||||
// fails with EISDIR. Treat it as absent, mirroring `size`.
|
||||
Err(e) if e.kind() == std::io::ErrorKind::IsADirectory => {
|
||||
Err(StorageError::NotFound)
|
||||
}
|
||||
Err(e) => Err(e.into()),
|
||||
}
|
||||
}
|
||||
@@ -108,7 +113,14 @@ impl Storage for LocalStorage {
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let size_bytes = file.metadata().await?.len();
|
||||
let meta = file.metadata().await?;
|
||||
// Opening a directory succeeds on Unix, but it isn't a stored blob:
|
||||
// its inode "size" is meaningless and ReaderStream would fail mid-read.
|
||||
// Treat it as absent, mirroring `size` and `get`.
|
||||
if !meta.is_file() {
|
||||
return Err(StorageError::NotFound);
|
||||
}
|
||||
let size_bytes = meta.len();
|
||||
// 64 KiB chunks: small enough that a few-MB page emits many frames
|
||||
// (so streaming is observable), large enough to keep syscalls cheap.
|
||||
let stream = ReaderStream::with_capacity(file, 64 * 1024);
|
||||
@@ -127,6 +139,23 @@ impl Storage for LocalStorage {
|
||||
}
|
||||
}
|
||||
|
||||
async fn rename(&self, from: &str, to: &str) -> Result<(), StorageError> {
|
||||
let from_path = self.resolve(from)?;
|
||||
let to_path = self.resolve(to)?;
|
||||
// Create the destination parent so a rename into a not-yet-existing
|
||||
// chapter directory succeeds. `from` and `to` share the storage
|
||||
// root (same filesystem), so this is a cheap atomic metadata move,
|
||||
// not a copy.
|
||||
if let Some(parent) = to_path.parent() {
|
||||
fs::create_dir_all(parent).await?;
|
||||
}
|
||||
match fs::rename(&from_path, &to_path).await {
|
||||
Ok(()) => Ok(()),
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => Err(StorageError::NotFound),
|
||||
Err(e) => Err(e.into()),
|
||||
}
|
||||
}
|
||||
|
||||
async fn exists(&self, key: &str) -> Result<bool, StorageError> {
|
||||
let path: &Path = &self.resolve(key)?;
|
||||
Ok(fs::try_exists(path).await?)
|
||||
@@ -227,6 +256,21 @@ mod tests {
|
||||
assert!(matches!(s.size("adir").await, Err(StorageError::NotFound)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_on_directory_is_not_found() {
|
||||
// A key resolving to a directory isn't a stored blob. `fs::read` on a
|
||||
// dir errors with EISDIR (not NotFound), and `File::open` on a dir
|
||||
// succeeds on Unix then streams garbage — both must surface NotFound.
|
||||
let dir = tempdir().unwrap();
|
||||
let s = LocalStorage::new(dir.path());
|
||||
std::fs::create_dir(dir.path().join("adir")).unwrap();
|
||||
assert!(matches!(s.get("adir").await, Err(StorageError::NotFound)));
|
||||
assert!(matches!(
|
||||
s.get_stream("adir").await.err(),
|
||||
Some(StorageError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_stream_writes_full_body_and_removes_temp_on_error() {
|
||||
use bytes::Bytes;
|
||||
@@ -273,6 +317,34 @@ mod tests {
|
||||
assert_eq!(entries, vec!["ok.bin"]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn rename_moves_blob_and_creates_destination_dirs() {
|
||||
let dir = tempdir().unwrap();
|
||||
let s = LocalStorage::new(dir.path());
|
||||
s.put("staging/up/0000.png", b"page-bytes").await.unwrap();
|
||||
|
||||
// Destination dir doesn't exist yet — rename must create it.
|
||||
s.rename("staging/up/0000.png", "mangas/m/chapters/c/pages/0001.png")
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert!(!s.exists("staging/up/0000.png").await.unwrap(), "source gone");
|
||||
assert_eq!(
|
||||
s.get("mangas/m/chapters/c/pages/0001.png").await.unwrap(),
|
||||
b"page-bytes"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn rename_missing_source_is_not_found() {
|
||||
let dir = tempdir().unwrap();
|
||||
let s = LocalStorage::new(dir.path());
|
||||
assert!(matches!(
|
||||
s.rename("staging/nope.png", "dest/x.png").await,
|
||||
Err(StorageError::NotFound)
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_stream_emits_multiple_chunks_for_large_files() {
|
||||
use futures_util::StreamExt as _;
|
||||
|
||||
@@ -82,6 +82,27 @@ pub trait Storage: Send + Sync {
|
||||
async fn delete(&self, key: &str) -> Result<(), StorageError>;
|
||||
async fn exists(&self, key: &str) -> Result<bool, StorageError>;
|
||||
|
||||
/// Move a blob from `from` to `to`, overwriting any existing blob at
|
||||
/// `to`. Returns `NotFound` if `from` doesn't exist. The chapter
|
||||
/// upload path uses this to promote a staged page to its final,
|
||||
/// chapter-scoped key once the chapter row (and thus its id) exists —
|
||||
/// so pages can be streamed to storage as their multipart parts
|
||||
/// arrive, without buffering the whole chapter in memory.
|
||||
///
|
||||
/// The default implementation streams `from` to `to` and deletes the
|
||||
/// source, so backends without a native move still satisfy the
|
||||
/// contract. LocalStorage overrides it with a filesystem rename (an
|
||||
/// atomic metadata op within a mount); a future S3Storage would
|
||||
/// override with a server-side copy + delete.
|
||||
async fn rename(&self, from: &str, to: &str) -> Result<(), StorageError> {
|
||||
use futures_util::StreamExt as _;
|
||||
let StreamingFile { stream, .. } = self.get_stream(from).await?;
|
||||
let mapped: PutByteStream<'_> = Box::pin(stream.map(|r| r.map_err(StorageError::Io)));
|
||||
self.put_stream(to, mapped).await?;
|
||||
self.delete(from).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Size in bytes of the blob at `key`, `NotFound` when it doesn't
|
||||
/// exist. Cheap metadata lookup (local: `fs::metadata`; a future
|
||||
/// `S3Storage`: HEAD object). Used by the cover-capture path and the
|
||||
|
||||
@@ -6,7 +6,12 @@
|
||||
//! whitelist with 415. Filename and extension never reach the storage
|
||||
//! key — we derive both from the sniffed type.
|
||||
|
||||
use axum::extract::multipart::Field;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::api::mangas::map_multipart_error;
|
||||
use crate::error::{AppError, AppResult};
|
||||
use crate::storage::Storage;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct UploadedImage {
|
||||
@@ -15,6 +20,100 @@ pub struct UploadedImage {
|
||||
pub ext: &'static str,
|
||||
}
|
||||
|
||||
/// A page image written to a temporary staging key during a chapter
|
||||
/// upload, awaiting promotion to its final chapter-scoped key once the
|
||||
/// chapter row (and thus its id) exists. Carries only the small metadata
|
||||
/// the caller needs — never the image bytes.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StagedImage {
|
||||
pub staging_key: String,
|
||||
pub mime: &'static str,
|
||||
pub ext: &'static str,
|
||||
pub size_bytes: i64,
|
||||
}
|
||||
|
||||
/// The staging-key prefix. Blobs left here by a failed or abandoned upload
|
||||
/// are orphans a future reaper can sweep; a successful upload renames every
|
||||
/// staged page out of this prefix.
|
||||
pub const STAGING_PREFIX: &str = "staging";
|
||||
|
||||
/// Upper bound on a multipart `metadata` JSON part, enforced as bytes arrive
|
||||
/// (via [`read_capped`]). Manga/chapter metadata is title + a few short lists +
|
||||
/// a description — kilobytes at most. Without this, `metadata` was the one
|
||||
/// remaining part read with the unbounded `Field::bytes()`, letting a client
|
||||
/// buffer up to the whole 200 MiB request body in memory as a single JSON blob
|
||||
/// before any validation ran. 64 KiB is generous headroom over any legitimate
|
||||
/// payload while keeping the worst case tiny.
|
||||
pub const MAX_METADATA_BYTES: usize = 64 * 1024;
|
||||
|
||||
/// Read one multipart image part and write it straight to a staging key,
|
||||
/// returning only its metadata. The per-file byte cap is enforced as bytes
|
||||
/// arrive, so an oversized part is rejected (413) without being fully
|
||||
/// buffered, and at most one page's bytes are held in memory at a time —
|
||||
/// the whole chapter is never buffered, unlike the previous
|
||||
/// read-all-then-persist path.
|
||||
pub async fn stage_image_part(
|
||||
storage: &dyn Storage,
|
||||
field: Field<'_>,
|
||||
upload_id: Uuid,
|
||||
seq: usize,
|
||||
max_size: usize,
|
||||
field_name: &str,
|
||||
) -> AppResult<StagedImage> {
|
||||
let bytes = read_capped(field, max_size, field_name).await?;
|
||||
// Reuse the shared sniff + whitelist check; it re-verifies the size cap
|
||||
// (already enforced above) and derives mime/ext from magic bytes.
|
||||
let img = parse_image(bytes, max_size, field_name)?;
|
||||
let staging_key = format!("{STAGING_PREFIX}/{}/{:04}.{}", upload_id.simple(), seq, img.ext);
|
||||
storage.put(&staging_key, &img.bytes).await?;
|
||||
Ok(StagedImage {
|
||||
staging_key,
|
||||
mime: img.mime,
|
||||
ext: img.ext,
|
||||
size_bytes: img.bytes.len() as i64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Read one multipart part fully into memory, enforcing the per-file byte cap
|
||||
/// as chunks arrive so an oversized part is rejected (413) **without being fully
|
||||
/// buffered**. Use for parts whose bytes the caller needs in hand — e.g. the
|
||||
/// cover image, which is written to a manga-scoped key. Page parts should prefer
|
||||
/// [`stage_image_part`], which streams straight to storage.
|
||||
///
|
||||
/// Replaces the `Field::bytes()` path, which buffered the entire field before
|
||||
/// any size check ran, letting an attacker allocate an arbitrarily large body
|
||||
/// before the cap kicked in.
|
||||
pub async fn read_capped(
|
||||
mut field: Field<'_>,
|
||||
max_size: usize,
|
||||
field_name: &str,
|
||||
) -> AppResult<Vec<u8>> {
|
||||
let mut bytes: Vec<u8> = Vec::new();
|
||||
while let Some(chunk) = field.chunk().await.map_err(map_multipart_error)? {
|
||||
push_capped(&mut bytes, &chunk, max_size, field_name)?;
|
||||
}
|
||||
Ok(bytes)
|
||||
}
|
||||
|
||||
/// Append `chunk` to `buf`, rejecting with 413 as soon as the running total
|
||||
/// would exceed `max_size` — so the oversized chunk is never copied in. Split
|
||||
/// out from the read loop so the cap logic is unit-testable without a live
|
||||
/// multipart field.
|
||||
fn push_capped(
|
||||
buf: &mut Vec<u8>,
|
||||
chunk: &[u8],
|
||||
max_size: usize,
|
||||
field_name: &str,
|
||||
) -> AppResult<()> {
|
||||
if buf.len().saturating_add(chunk.len()) > max_size {
|
||||
return Err(AppError::PayloadTooLarge(format!(
|
||||
"{field_name} exceeds {max_size}-byte cap"
|
||||
)));
|
||||
}
|
||||
buf.extend_from_slice(chunk);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn parse_image(bytes: Vec<u8>, max_size: usize, field_name: &str) -> AppResult<UploadedImage> {
|
||||
if bytes.len() > max_size {
|
||||
return Err(AppError::PayloadTooLarge(format!(
|
||||
@@ -114,4 +213,30 @@ mod tests {
|
||||
assert!(matches!(err, AppError::PayloadTooLarge(_)));
|
||||
assert_eq!(err.code(), "payload_too_large");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn push_capped_accumulates_under_cap() {
|
||||
let mut buf = Vec::new();
|
||||
push_capped(&mut buf, b"hello ", 100, "cover").unwrap();
|
||||
push_capped(&mut buf, b"world", 100, "cover").unwrap();
|
||||
assert_eq!(buf, b"hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn push_capped_rejects_before_copying_oversized_chunk() {
|
||||
// The whole point: the offending chunk must NOT be appended — an
|
||||
// oversized part is rejected without buffering it.
|
||||
let mut buf = vec![0u8; 90];
|
||||
let err = push_capped(&mut buf, &[0u8; 20], 100, "cover").unwrap_err();
|
||||
assert!(matches!(err, AppError::PayloadTooLarge(_)));
|
||||
assert_eq!(err.code(), "payload_too_large");
|
||||
assert_eq!(buf.len(), 90, "buffer must not grow past the cap");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn push_capped_allows_exactly_at_cap() {
|
||||
let mut buf = vec![0u8; 90];
|
||||
push_capped(&mut buf, &[0u8; 10], 100, "cover").unwrap();
|
||||
assert_eq!(buf.len(), 100);
|
||||
}
|
||||
}
|
||||
|
||||
140
backend/tests/analysis_ocr.rs
Normal file
140
backend/tests/analysis_ocr.rs
Normal file
@@ -0,0 +1,140 @@
|
||||
//! Integration tests for the OCR analysis backend
|
||||
//! (`analysis::ocr::OcrAnalyzeDispatcher`). A stub OCR engine stands in for
|
||||
//! `ocrs` (whose `.rten` models aren't shipped to CI), so these pin the
|
||||
//! storage→OCR→persist wiring: the dispatcher reads the page image, runs the
|
||||
//! engine, and persists the lines via the shared `persist_analysis` path —
|
||||
//! landing `page_ocr_text` rows and a populated `search_doc` exactly like the
|
||||
//! vision backend. Each `#[sqlx::test]` gets a fresh migrated DB.
|
||||
|
||||
mod common;
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use mangalord::analysis::daemon::AnalyzeDispatcher;
|
||||
use mangalord::analysis::ocr::test_support::StubOcrEngine;
|
||||
use mangalord::analysis::ocr::OcrAnalyzeDispatcher;
|
||||
use mangalord::domain::page_analysis::AnalysisStatus;
|
||||
use mangalord::repo;
|
||||
use mangalord::storage::{LocalStorage, Storage};
|
||||
use sqlx::PgPool;
|
||||
use tempfile::TempDir;
|
||||
use uuid::Uuid;
|
||||
|
||||
/// Seed a manga → chapter → page chain whose page points at `storage_key`,
|
||||
/// and return the page id.
|
||||
async fn seed_page(pool: &PgPool, storage_key: &str) -> Uuid {
|
||||
let manga_id: Uuid =
|
||||
sqlx::query_scalar("INSERT INTO mangas (title) VALUES ('M') RETURNING id")
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
.unwrap();
|
||||
let chapter_id: Uuid = sqlx::query_scalar(
|
||||
"INSERT INTO chapters (manga_id, number) VALUES ($1, 1) RETURNING id",
|
||||
)
|
||||
.bind(manga_id)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
.unwrap();
|
||||
sqlx::query_scalar(
|
||||
"INSERT INTO pages (chapter_id, page_number, storage_key, content_type) \
|
||||
VALUES ($1, 1, $2, 'image/png') RETURNING id",
|
||||
)
|
||||
.bind(chapter_id)
|
||||
.bind(storage_key)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn ocr_dispatcher(
|
||||
pool: &PgPool,
|
||||
storage: Arc<dyn Storage>,
|
||||
lines: &[&str],
|
||||
) -> OcrAnalyzeDispatcher {
|
||||
OcrAnalyzeDispatcher {
|
||||
db: pool.clone(),
|
||||
storage,
|
||||
engine: StubOcrEngine::new(lines),
|
||||
max_image_bytes: 8 * 1024 * 1024,
|
||||
ocr_permits: Arc::new(tokio::sync::Semaphore::new(1)),
|
||||
}
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dispatch_persists_ocr_lines_and_search_doc(pool: PgPool) {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let storage: Arc<dyn Storage> = Arc::new(LocalStorage::new(dir.path()));
|
||||
let key = "mangas/x/p1.png";
|
||||
storage.put(key, &common::fake_png_bytes()).await.unwrap();
|
||||
let page_id = seed_page(&pool, key).await;
|
||||
|
||||
let dispatcher = ocr_dispatcher(&pool, Arc::clone(&storage), &["Hello there", "general"]);
|
||||
dispatcher.dispatch(page_id).await.unwrap();
|
||||
|
||||
// Two OCR rows, in order, with the recognized text.
|
||||
let rows: Vec<(String, i32)> = sqlx::query_as(
|
||||
"SELECT text, ord FROM page_ocr_text WHERE page_id = $1 ORDER BY ord",
|
||||
)
|
||||
.bind(page_id)
|
||||
.fetch_all(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(rows.len(), 2);
|
||||
assert_eq!(rows[0].0, "Hello there");
|
||||
assert_eq!(rows[1].0, "general");
|
||||
|
||||
// The analysis row is `done`, stamped with the ocrs model label, and has a
|
||||
// non-empty tsvector so text search works.
|
||||
let row = repo::page_analysis::load(&pool, page_id).await.unwrap().unwrap();
|
||||
assert_eq!(row.status, AnalysisStatus::Done);
|
||||
assert_eq!(row.model.as_deref(), Some("ocrs"));
|
||||
let has_doc: bool = sqlx::query_scalar(
|
||||
"SELECT search_doc IS NOT NULL AND search_doc != ''::tsvector \
|
||||
FROM page_analysis WHERE page_id = $1",
|
||||
)
|
||||
.bind(page_id)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(has_doc, "search_doc must be populated from OCR text");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dispatch_rejects_page_image_over_the_byte_cap(pool: PgPool) {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let storage: Arc<dyn Storage> = Arc::new(LocalStorage::new(dir.path()));
|
||||
let key = "mangas/x/big.png";
|
||||
// 4 KiB on disk, 1 KiB cap — the streamed read must bail on the cap
|
||||
// rather than buffering the whole blob and OCR-ing it.
|
||||
storage.put(key, &vec![0u8; 4096]).await.unwrap();
|
||||
let page_id = seed_page(&pool, key).await;
|
||||
|
||||
let dispatcher = OcrAnalyzeDispatcher {
|
||||
db: pool.clone(),
|
||||
storage: Arc::clone(&storage),
|
||||
engine: StubOcrEngine::new(&["should not run"]),
|
||||
max_image_bytes: 1024,
|
||||
ocr_permits: Arc::new(tokio::sync::Semaphore::new(1)),
|
||||
};
|
||||
let err = dispatcher.dispatch(page_id).await.unwrap_err();
|
||||
assert!(
|
||||
err.chain().any(|c| c.to_string().contains("cap")),
|
||||
"expected an over-cap error, got: {err:#}"
|
||||
);
|
||||
|
||||
// A rejected page must not land an analysis row.
|
||||
assert!(repo::page_analysis::load(&pool, page_id)
|
||||
.await
|
||||
.unwrap()
|
||||
.is_none());
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dispatch_missing_page_is_noop(pool: PgPool) {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let storage: Arc<dyn Storage> = Arc::new(LocalStorage::new(dir.path()));
|
||||
// A page id that was never inserted — the dispatcher must treat it as a
|
||||
// deleted page and succeed without writing anything.
|
||||
let dispatcher = ocr_dispatcher(&pool, storage, &["whatever"]);
|
||||
dispatcher.dispatch(Uuid::new_v4()).await.unwrap();
|
||||
}
|
||||
@@ -74,6 +74,38 @@ async fn analyze_job_count(pool: &PgPool) -> i64 {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn force_analyze_enqueue_rolls_back_with_its_transaction(pool: PgPool) {
|
||||
// The admin force-reanalyze handler runs enqueue_for_page_conn + the audit
|
||||
// insert in one transaction. Prove the enqueue genuinely participates in
|
||||
// that tx: when the tx is abandoned (the path taken if the audit insert
|
||||
// fails and the `?` propagates), no job survives — so the audit trail can
|
||||
// never miss a force-reanalyze that actually landed.
|
||||
let page_id = seed_pages(&pool, 1).await[0];
|
||||
|
||||
let mut tx = pool.begin().await.unwrap();
|
||||
let outcome =
|
||||
mangalord::repo::page_analysis::enqueue_for_page_conn(&mut tx, page_id, true)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(matches!(
|
||||
outcome,
|
||||
mangalord::repo::page_analysis::EnqueueForPageOutcome::Inserted
|
||||
));
|
||||
// Visible inside the open transaction…
|
||||
let in_tx: i64 = sqlx::query_scalar(
|
||||
"SELECT COUNT(*) FROM crawler_jobs WHERE payload->>'kind' = 'analyze_page'",
|
||||
)
|
||||
.fetch_one(&mut *tx)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(in_tx, 1);
|
||||
|
||||
// …but gone once the tx rolls back instead of committing.
|
||||
tx.rollback().await.unwrap();
|
||||
assert_eq!(analyze_job_count(&pool).await, 0);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn chapter_upload_enqueues_one_analysis_job_per_page(pool: PgPool) {
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
@@ -157,6 +189,51 @@ async fn reenqueue_returns_503_when_analysis_disabled(pool: PgPool) {
|
||||
assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reenqueue_rejects_a_malformed_body(pool: PgPool) {
|
||||
// A present-but-malformed JSON body must be a 422 — NOT silently coerced to
|
||||
// an absent body, which would run the default full "All" scope the caller
|
||||
// never asked for and enqueue jobs across the whole library.
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
let cookie = seed_admin(&pool, &h.app).await;
|
||||
seed_pages(&pool, 2).await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_raw_with_cookie(
|
||||
"/api/v1/admin/analysis/reenqueue",
|
||||
"{ not valid json",
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::UNPROCESSABLE_ENTITY);
|
||||
assert_eq!(analyze_job_count(&pool).await, 0, "a rejected body enqueues nothing");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reenqueue_treats_empty_body_as_default_all_scope(pool: PgPool) {
|
||||
// An absent/empty body is still valid: it means the default "All" scope.
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
let cookie = seed_admin(&pool, &h.app).await;
|
||||
seed_pages(&pool, 2).await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_raw_with_cookie(
|
||||
"/api/v1/admin/analysis/reenqueue",
|
||||
"",
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["enqueued"], 2, "empty body backfills the whole library");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reenqueue_backfills_existing_pages(pool: PgPool) {
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
@@ -194,6 +271,38 @@ async fn reenqueue_backfills_existing_pages(pool: PgPool) {
|
||||
assert_eq!(analyze_job_count(&pool).await, 3);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reenqueue_writes_audit_row_atomically_with_jobs(pool: PgPool) {
|
||||
// The enqueue and its admin_audit row are committed in one transaction,
|
||||
// so a successful re-enqueue always leaves both the jobs AND exactly one
|
||||
// matching audit row — the audit trail can't miss an enqueue that landed.
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
let cookie = seed_admin(&pool, &h.app).await;
|
||||
seed_pages(&pool, 2).await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/admin/analysis/reenqueue",
|
||||
json!({ "only_unanalyzed": true }),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
assert_eq!(analyze_job_count(&pool).await, 2);
|
||||
|
||||
let audit: serde_json::Value = sqlx::query_scalar(
|
||||
"SELECT payload FROM admin_audit WHERE action = 'analysis_reenqueue'",
|
||||
)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.expect("exactly one analysis_reenqueue audit row committed with the jobs");
|
||||
assert_eq!(audit["enqueued"], 2);
|
||||
assert_eq!(audit["only_unanalyzed"], true);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reenqueue_scoped_to_manga(pool: PgPool) {
|
||||
let h = common::harness_with_analysis(pool.clone());
|
||||
|
||||
@@ -279,7 +279,7 @@ async fn backfill_fills_unmeasured_pages_and_covers_idempotently(pool: PgPool) {
|
||||
let body2 = common::body_json(resp2).await;
|
||||
assert_eq!(body2["pages"].as_i64().unwrap(), 0);
|
||||
assert_eq!(body2["covers"].as_i64().unwrap(), 0);
|
||||
assert_eq!(body2["more_remaining"].as_bool().unwrap(), false);
|
||||
assert!(!body2["more_remaining"].as_bool().unwrap());
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
@@ -362,7 +362,7 @@ async fn backfill_caps_per_run_and_reports_more_remaining(pool: PgPool) {
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
// Capped at 20_000 attempts this run; one row left → run again.
|
||||
assert_eq!(body["more_remaining"].as_bool().unwrap(), true);
|
||||
assert!(body["more_remaining"].as_bool().unwrap());
|
||||
assert_eq!(body["missing"].as_i64().unwrap(), 20_000);
|
||||
assert_eq!(body["pages"].as_i64().unwrap(), 0);
|
||||
}
|
||||
@@ -403,9 +403,8 @@ async fn backfill_at_exact_cap_is_not_more_remaining(pool: PgPool) {
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["missing"].as_i64().unwrap(), 20_000);
|
||||
assert_eq!(
|
||||
body["more_remaining"].as_bool().unwrap(),
|
||||
false,
|
||||
assert!(
|
||||
!body["more_remaining"].as_bool().unwrap(),
|
||||
"exactly-cap backlog drains in one run; no follow-up needed"
|
||||
);
|
||||
}
|
||||
|
||||
@@ -147,6 +147,46 @@ async fn list_filters_by_substring_search(pool: PgPool) {
|
||||
assert_eq!(body["page"]["total"], 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn search_treats_like_wildcards_literally(pool: PgPool) {
|
||||
let h = common::harness(pool.clone());
|
||||
let (_admin_name, cookie, _) = seed_admin(&pool, &h.app).await;
|
||||
// Two usernames differing only at one position (underscores are legal in
|
||||
// usernames). `_` is a LIKE single-char wildcard: unescaped, `%a_b%` matches
|
||||
// BOTH; escaped, only the literal "a_b" username. Admin user search has no
|
||||
// trigram OR, so length/similarity don't matter here.
|
||||
for username in ["axbfindme", "a_bfindme"] {
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json(
|
||||
"/api/v1/auth/register",
|
||||
json!({ "username": username, "password": "hunter2hunter2" }),
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::CREATED);
|
||||
}
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get_with_cookie(
|
||||
"/api/v1/admin/users?search=a_b",
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
let items = body["items"].as_array().unwrap();
|
||||
assert_eq!(
|
||||
items.len(),
|
||||
1,
|
||||
"the `_` must match literally: only a_bfindme, not axbfindme"
|
||||
);
|
||||
assert_eq!(items[0]["username"], "a_bfindme");
|
||||
}
|
||||
|
||||
// ---- self-protection -------------------------------------------------------
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
@@ -381,10 +421,11 @@ async fn delete_writes_audit_row(pool: PgPool) {
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::NO_CONTENT);
|
||||
|
||||
let rows: Vec<(Option<Uuid>, String, String, Option<Uuid>, serde_json::Value)> =
|
||||
sqlx::query_as(
|
||||
"SELECT actor_user_id, action, target_kind, target_id, payload FROM admin_audit",
|
||||
)
|
||||
// (actor_user_id, action, target_kind, target_id, payload)
|
||||
type AuditRow = (Option<Uuid>, String, String, Option<Uuid>, serde_json::Value);
|
||||
let rows: Vec<AuditRow> = sqlx::query_as(
|
||||
"SELECT actor_user_id, action, target_kind, target_id, payload FROM admin_audit",
|
||||
)
|
||||
.fetch_all(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
@@ -9,6 +9,73 @@ fn creds(username: &str) -> serde_json::Value {
|
||||
json!({ "username": username, "password": "hunter2hunter2" })
|
||||
}
|
||||
|
||||
/// A login request carrying a chosen `X-Forwarded-For`. Empty password so the
|
||||
/// handler consumes a rate-limit token then short-circuits with 400 before any
|
||||
/// argon2/DB work — the burst drains near-instantly regardless of CI load.
|
||||
fn login_from(xff: &str) -> axum::http::Request<axum::body::Body> {
|
||||
axum::http::Request::builder()
|
||||
.method("POST")
|
||||
.uri("/api/v1/auth/login")
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("x-forwarded-for", xff)
|
||||
.body(axum::body::Body::from(
|
||||
json!({ "username": "victim", "password": "" }).to_string(),
|
||||
))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
// The per-IP rate limiter's soundness hinges on one gate: X-Forwarded-For is
|
||||
// honored ONLY when AUTH_TRUSTED_PROXY is set. These two tests pin both sides of
|
||||
// that gate at the request level (the pure parser is unit-tested separately).
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn trusted_proxy_gives_each_forwarded_ip_its_own_bucket(pool: PgPool) {
|
||||
let h = common::harness_with_auth_rate_limit_proxy(pool, 1, 2, true);
|
||||
// Drain IP A's bucket (per_sec=1, burst=2) until it 429s.
|
||||
let mut a_saw_429 = false;
|
||||
for _ in 0..8 {
|
||||
let resp = h.app.clone().oneshot(login_from("203.0.113.10")).await.unwrap();
|
||||
if resp.status() == StatusCode::TOO_MANY_REQUESTS {
|
||||
a_saw_429 = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
assert!(a_saw_429, "IP A must be rate-limited after draining its own bucket");
|
||||
// A different forwarded IP has an independent bucket — its first hit is not 429.
|
||||
let resp_b = h.app.clone().oneshot(login_from("198.51.100.20")).await.unwrap();
|
||||
assert_ne!(
|
||||
resp_b.status(),
|
||||
StatusCode::TOO_MANY_REQUESTS,
|
||||
"a distinct X-Forwarded-For hop must get its own bucket, not A's drained one"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn untrusted_proxy_ignores_forwarded_ip_and_shares_one_bucket(pool: PgPool) {
|
||||
let h = common::harness_with_auth_rate_limit_proxy(pool, 1, 2, false);
|
||||
// Every request carries a DISTINCT (spoofed) X-Forwarded-For, but the backend
|
||||
// doesn't trust it — so they all fall back to the single global bucket, which
|
||||
// drains and starts returning 429. If XFF were wrongly honored here, each
|
||||
// unique IP would get a fresh bucket and none of these would ever 429.
|
||||
let mut saw_429 = false;
|
||||
for i in 0..12 {
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(login_from(&format!("10.0.0.{i}")))
|
||||
.await
|
||||
.unwrap();
|
||||
if resp.status() == StatusCode::TOO_MANY_REQUESTS {
|
||||
saw_429 = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
assert!(
|
||||
saw_429,
|
||||
"with trusted_proxy off, spoofed X-Forwarded-For must NOT dodge the shared limiter"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn register_creates_user_and_sets_session_cookie(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
@@ -170,6 +237,26 @@ async fn login_rejects_wrong_password(pool: PgPool) {
|
||||
assert_eq!(body["error"]["code"], "unauthenticated");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn login_rejects_oversized_password_before_argon2(pool: PgPool) {
|
||||
// A multi-KB password on the login path must be rejected as malformed
|
||||
// input (400) rather than fed to argon2 — otherwise every attempt is a
|
||||
// CPU-DoS. The account need not even exist; the guard is input-shape only.
|
||||
let h = common::harness(pool);
|
||||
let giant = "a".repeat(5000);
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::post_json(
|
||||
"/api/v1/auth/login",
|
||||
json!({ "username": "alice", "password": giant }),
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::BAD_REQUEST);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["error"]["code"], "invalid_input");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn login_rejects_unknown_user(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
@@ -521,6 +608,156 @@ async fn create_and_use_bot_token(pool: PgPool) {
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_tokens_returns_callers_tokens_scoped_and_without_hash(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
|
||||
// Mint two tokens for this user, one with an expiry.
|
||||
for body in [
|
||||
json!({ "name": "no-expiry" }),
|
||||
json!({ "name": "expiring", "expires_in_days": 30 }),
|
||||
] {
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/auth/tokens",
|
||||
body,
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::CREATED);
|
||||
}
|
||||
|
||||
// A second user's token must NOT appear in the first user's list.
|
||||
let (_, other) = common::register_user(&h.app).await;
|
||||
let _ = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/auth/tokens",
|
||||
json!({ "name": "someone-elses" }),
|
||||
&other,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get_with_cookie("/api/v1/auth/tokens", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
let items = body["items"].as_array().unwrap();
|
||||
assert_eq!(items.len(), 2, "only the caller's two tokens");
|
||||
|
||||
let names: Vec<&str> = items.iter().map(|t| t["name"].as_str().unwrap()).collect();
|
||||
assert!(names.contains(&"no-expiry") && names.contains(&"expiring"));
|
||||
// Raw secret / hash must never appear, but expiry metadata must.
|
||||
for t in items {
|
||||
assert!(t.get("token_hash").is_none(), "token_hash must be absent");
|
||||
assert!(t.get("bearer").is_none(), "raw bearer only shown at creation");
|
||||
assert!(t.get("expires_at").is_some(), "expiry metadata present");
|
||||
}
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn bot_token_with_future_expiry_authenticates(pool: PgPool) {
|
||||
// A token minted with expires_in_days is still active before its
|
||||
// expiry, and the response echoes a non-null expires_at.
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/auth/tokens",
|
||||
json!({ "name": "ci-bot", "expires_in_days": 30 }),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::CREATED);
|
||||
let body = common::body_json(resp).await;
|
||||
assert!(
|
||||
body["expires_at"].is_string(),
|
||||
"expires_at should be set, got {}",
|
||||
body["expires_at"]
|
||||
);
|
||||
let bearer = body["bearer"].as_str().unwrap().to_string();
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get_with_bearer("/api/v1/auth/me", &bearer))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn expired_bot_token_is_rejected(pool: PgPool) {
|
||||
use chrono::{Duration, Utc};
|
||||
use mangalord::auth::token::generate_token;
|
||||
|
||||
let h = common::harness(pool.clone());
|
||||
common::register_user(&h.app).await;
|
||||
let user_id: uuid::Uuid = sqlx::query_scalar("SELECT id FROM users LIMIT 1")
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Hand-craft a token that expired an hour ago.
|
||||
let (raw, hash) = generate_token();
|
||||
let expires_at = Utc::now() - Duration::hours(1);
|
||||
sqlx::query(
|
||||
"INSERT INTO api_tokens (user_id, name, token_hash, expires_at) \
|
||||
VALUES ($1, 'stale', $2, $3)",
|
||||
)
|
||||
.bind(user_id)
|
||||
.bind(&hash[..])
|
||||
.bind(expires_at)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get_with_bearer("/api/v1/auth/me", &raw))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::UNAUTHORIZED);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["error"]["code"], "unauthenticated");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn create_token_rejects_out_of_range_expiry(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
|
||||
for days in [0, -5, 100_000] {
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/auth/tokens",
|
||||
json!({ "name": "bad", "expires_in_days": days }),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
resp.status(),
|
||||
StatusCode::UNPROCESSABLE_ENTITY,
|
||||
"expires_in_days={days} should be rejected"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn user_a_cannot_delete_user_b_token(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
|
||||
@@ -30,6 +30,47 @@ fn first_author_id(manga: &Value) -> String {
|
||||
manga["authors"][0]["id"].as_str().unwrap().to_string()
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn search_treats_like_wildcards_literally(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
// Long author names differing only at one position, short search term — so
|
||||
// the trigram OR (which keeps the raw term by design) stays under threshold
|
||||
// and the ILIKE branch is what decides. Unescaped `%a_b%` matches both the
|
||||
// "axb" and "a_b" names; escaped, only the literal "a_b" name.
|
||||
create_manga(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "title": "M1", "authors": ["The Quick Brown Fox axb Jumps Over"] }),
|
||||
)
|
||||
.await;
|
||||
create_manga(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "title": "M2", "authors": ["The Quick Brown Fox a_b Jumps Over"] }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get("/api/v1/authors?search=a_b"))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
let names: Vec<&str> = body
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|a| a["name"].as_str().unwrap())
|
||||
.collect();
|
||||
assert_eq!(
|
||||
names,
|
||||
vec!["The Quick Brown Fox a_b Jumps Over"],
|
||||
"the `_` in the search term must match literally, not as a wildcard"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_returns_name_and_manga_count(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
|
||||
@@ -52,7 +52,7 @@ async fn create_then_list_returns_only_own(pool: PgPool) {
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn create_returns_409_on_duplicate_manga_level(pool: PgPool) {
|
||||
async fn create_is_idempotent_on_duplicate_manga_level(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
@@ -64,12 +64,26 @@ async fn create_returns_409_on_duplicate_manga_level(pool: PgPool) {
|
||||
&cookie,
|
||||
)
|
||||
};
|
||||
// First add creates (201); re-adding is an idempotent no-op (200) that
|
||||
// returns the SAME bookmark rather than a 409 — collections behave this way
|
||||
// too, so the UI shouldn't surface a false "Could not add bookmark" error.
|
||||
let first = h.app.clone().oneshot(make()).await.unwrap();
|
||||
assert_eq!(first.status(), StatusCode::CREATED);
|
||||
let second = h.app.oneshot(make()).await.unwrap();
|
||||
assert_eq!(second.status(), StatusCode::CONFLICT);
|
||||
let body = common::body_json(second).await;
|
||||
assert_eq!(body["error"]["code"], "conflict");
|
||||
let first_body = common::body_json(first).await;
|
||||
|
||||
let second = h.app.clone().oneshot(make()).await.unwrap();
|
||||
assert_eq!(second.status(), StatusCode::OK);
|
||||
let second_body = common::body_json(second).await;
|
||||
assert_eq!(second_body["id"], first_body["id"], "same bookmark returned");
|
||||
|
||||
// Exactly one row exists.
|
||||
let list = h
|
||||
.app
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/bookmarks", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(list).await;
|
||||
assert_eq!(body["items"].as_array().unwrap().len(), 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
@@ -225,13 +239,16 @@ async fn concurrent_manga_bookmarks_serialised_by_unique_index(pool: PgPool) {
|
||||
|
||||
let (s1, s2) = tokio::join!(f1, f2);
|
||||
let statuses = [s1.unwrap(), s2.unwrap()];
|
||||
// The unique index serialises the two inserts: one wins with 201, the other
|
||||
// hits the violation and — now idempotent — returns the existing bookmark
|
||||
// with 200 rather than a 409.
|
||||
assert!(
|
||||
statuses.contains(&StatusCode::CREATED),
|
||||
"expected one winner with 201, got {statuses:?}"
|
||||
);
|
||||
assert!(
|
||||
statuses.contains(&StatusCode::CONFLICT),
|
||||
"expected one loser with 409 (the partial unique index), got {statuses:?}"
|
||||
statuses.contains(&StatusCode::OK),
|
||||
"expected the loser to return 200 (idempotent), got {statuses:?}"
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -8,6 +8,8 @@ use uuid::Uuid;
|
||||
#[allow(unused_imports)]
|
||||
use serde_json as _;
|
||||
|
||||
use common::MultipartBuilder;
|
||||
|
||||
async fn seed_manga(h: &common::Harness, cookie: &str, title: &str) -> Uuid {
|
||||
common::seed_manga_via_api(&h.app, cookie, title).await
|
||||
}
|
||||
@@ -146,6 +148,49 @@ async fn list_chapters_returned_in_number_order(pool: PgPool) {
|
||||
assert_eq!(body["items"][1]["title"], serde_json::Value::Null);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_interleaves_uploaded_and_crawled_chapters_by_number(pool: PgPool) {
|
||||
// Mixed source: crawled chapters carry a `source_index`; an uploaded
|
||||
// chapter has none. The uploaded chapter must interleave by NUMBER, not sort
|
||||
// after every crawled chapter (the old `source_index DESC NULLS LAST` bug
|
||||
// dumped uploaded chapter 2 to the end, giving [1, 3, 2]).
|
||||
let h = common::harness(pool.clone());
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = seed_manga(&h, &cookie, "Berserk").await;
|
||||
|
||||
// Two crawled chapters (1 and 3) with DOM positions — newest-first, so
|
||||
// chapter 3 is at source_index 0 and chapter 1 at source_index 1.
|
||||
let c1 = seed_chapter(&pool, manga_id, 1, Some("Crawled One")).await;
|
||||
let c3 = seed_chapter(&pool, manga_id, 3, Some("Crawled Three")).await;
|
||||
sqlx::query("UPDATE chapters SET source_index = 1 WHERE id = $1")
|
||||
.bind(c1)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
sqlx::query("UPDATE chapters SET source_index = 0 WHERE id = $1")
|
||||
.bind(c3)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
// Uploaded chapter 2 — no source_index.
|
||||
seed_chapter(&pool, manga_id, 2, Some("Uploaded Two")).await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get(&format!("/api/v1/mangas/{manga_id}/chapters")))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
let numbers: Vec<i64> = body["items"]
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|c| c["number"].as_i64().unwrap())
|
||||
.collect();
|
||||
assert_eq!(numbers, vec![1, 2, 3], "uploaded chapter 2 must slot between 1 and 3");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_chapters_returns_404_for_unknown_manga(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
@@ -272,3 +317,68 @@ async fn list_pages_returns_404_for_unknown_chapter(pool: PgPool) {
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::NOT_FOUND);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn non_owner_can_upload_chapter(pool: PgPool) {
|
||||
// Contract lock: chapter upload is INTENTIONALLY open to any
|
||||
// authenticated user, not just the manga's creator. A user who did not
|
||||
// create the manga can still contribute a chapter (community
|
||||
// contributions), unlike manga-record edits which gate on ownership.
|
||||
// If this ever needs to become owner-only, that's a deliberate change —
|
||||
// this test (and the doc comment on `api::chapters::create`) should be
|
||||
// updated together, not silently broken.
|
||||
let h = common::harness(pool);
|
||||
|
||||
// Owner creates the manga.
|
||||
let (_owner, owner_cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = seed_manga(&h, &owner_cookie, "Berserk").await;
|
||||
|
||||
// A different, non-owner user uploads a chapter to it.
|
||||
let (_other, other_cookie) = common::register_user(&h.app).await;
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_multipart_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/chapters"),
|
||||
MultipartBuilder::new()
|
||||
.add_json("metadata", json!({ "number": 1, "title": "Contributed" }))
|
||||
.add_file("page", "1.png", "image/png", &common::fake_png_bytes()),
|
||||
&other_cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
resp.status(),
|
||||
StatusCode::CREATED,
|
||||
"a non-owner authenticated user must be able to upload a chapter"
|
||||
);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["number"], 1);
|
||||
assert_eq!(body["title"], "Contributed");
|
||||
assert_eq!(body["page_count"], 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn chapter_upload_rejects_oversized_metadata_part(pool: PgPool) {
|
||||
// The chapter `metadata` JSON part is capped as bytes arrive, just like the
|
||||
// manga one, so a client can't buffer a huge JSON blob in memory. A ~200 KiB
|
||||
// title blows the metadata cap before any page is staged.
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = seed_manga(&h, &cookie, "Berserk").await;
|
||||
|
||||
let huge = "A".repeat(200 * 1024);
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::post_multipart_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/chapters"),
|
||||
MultipartBuilder::new()
|
||||
.add_json("metadata", json!({ "number": 1, "title": huge }))
|
||||
.add_file("page", "1.png", "image/png", &common::fake_png_bytes()),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::PAYLOAD_TOO_LARGE);
|
||||
}
|
||||
|
||||
74
backend/tests/api_files.rs
Normal file
74
backend/tests/api_files.rs
Normal file
@@ -0,0 +1,74 @@
|
||||
mod common;
|
||||
|
||||
use axum::http::StatusCode;
|
||||
use tower::ServiceExt;
|
||||
|
||||
/// Write a blob straight into the harness storage root at `key`, bypassing the
|
||||
/// upload handlers — the only way to land a key whose extension resolves to the
|
||||
/// `application/octet-stream` fallback (uploads always mint image extensions).
|
||||
fn write_blob(h: &common::Harness, key: &str, bytes: &[u8]) {
|
||||
let path = h._storage_dir.path().join(key);
|
||||
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
std::fs::write(path, bytes).unwrap();
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn octet_stream_blobs_are_served_as_attachment(pool: sqlx::PgPool) {
|
||||
// A blob with an unknown extension serves as application/octet-stream. Such a
|
||||
// body could be crafted HTML/JS, so it must never render inline: force a
|
||||
// download with Content-Disposition: attachment (nosniff is already set).
|
||||
let h = common::harness(pool);
|
||||
write_blob(&h, "misc/blob.bin", b"\x00\x01not-an-image");
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get("/api/v1/files/misc/blob.bin"))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
assert_eq!(
|
||||
resp.headers().get("content-type").unwrap(),
|
||||
"application/octet-stream"
|
||||
);
|
||||
assert_eq!(
|
||||
resp.headers().get("content-disposition").unwrap(),
|
||||
"attachment"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn image_blobs_are_served_inline(pool: sqlx::PgPool) {
|
||||
// Regression guard: known image types keep rendering inline (no attachment
|
||||
// disposition), so covers/pages still display in the reader.
|
||||
let h = common::harness(pool);
|
||||
write_blob(&h, "misc/pic.png", &common::fake_png_bytes());
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get("/api/v1/files/misc/pic.png"))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
assert_eq!(resp.headers().get("content-type").unwrap(), "image/png");
|
||||
assert!(
|
||||
resp.headers().get("content-disposition").is_none(),
|
||||
"images must render inline, not download"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn corrupt_image_thumbnail_falls_back_to_original_not_500(pool: sqlx::PgPool) {
|
||||
// A blob with valid PNG magic bytes but an undecodable body passes upload's
|
||||
// magic-byte sniff; a ?w= thumbnail request then tries to decode it. That
|
||||
// must not 500 the reader — fall back to streaming the original bytes.
|
||||
let h = common::harness(pool);
|
||||
write_blob(&h, "misc/corrupt.png", &common::fake_png_bytes());
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get("/api/v1/files/misc/corrupt.png?w=320"))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK, "corrupt thumbnail must not 500");
|
||||
assert_eq!(resp.headers().get("content-type").unwrap(), "image/png");
|
||||
}
|
||||
@@ -164,6 +164,199 @@ async fn list_is_per_user_only(pool: PgPool) {
|
||||
assert_eq!(body["items"], json!([]));
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_reports_new_chapters_since_last_read(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
// Three chapters exist; the reader is on chapter 1, so chapters 2 and
|
||||
// 3 are "new since last read".
|
||||
let ch1 = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 3).await;
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "chapter_id": ch1, "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/read-progress", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"][0]["new_chapters_count"], 2);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_reports_last_read_chapter_page_count(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
// seed_chapter uploads a single page, so page_count is 1.
|
||||
let ch1 = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "chapter_id": ch1, "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/read-progress", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
// Exposes the last-read chapter's page count so a client can tell a
|
||||
// finished series (on the last page) from one still in progress.
|
||||
assert_eq!(body["items"][0]["chapter_page_count"], 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_new_chapters_count_is_zero_at_latest_chapter(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let ch2 = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
// Caught up to the last chapter → nothing newer.
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "chapter_id": ch2, "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/read-progress", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"][0]["new_chapters_count"], 0);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_new_chapters_count_is_zero_without_a_read_chapter(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
// Progress recorded without a chapter (manga-level) — we can't place
|
||||
// the reader among the chapters, so we don't claim any are "new".
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/read-progress", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"][0]["new_chapters_count"], 0);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_new_chapters_count_dedupes_same_numbered_scanlations(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
// (manga_id, number) is non-unique — two scanlations of chapter 2.
|
||||
// "New since last read" should count distinct chapter numbers past the
|
||||
// reader (2, 3 → 2), not raw rows (2, 2, 3 → 3).
|
||||
let ch1 = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 3).await;
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "chapter_id": ch1, "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/read-progress", &cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"][0]["new_chapters_count"], 2);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_single_manga_reports_new_chapters_count(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
let ch1 = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
// Duplicate scanlation of chapter 3 must not inflate the count.
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 3).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 3).await;
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "chapter_id": ch1, "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/me/read-progress/{manga_id}"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
// Distinct numbers past chapter 1: {2, 3} → 2.
|
||||
assert_eq!(body["new_chapters_count"], 2);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_single_manga_new_chapters_count_zero_without_read_chapter(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 1).await;
|
||||
let _ = seed_chapter(&h.app, &cookie, manga_id, 2).await;
|
||||
// Manga-level progress (no chapter) → position unknown → 0.
|
||||
let _ = upsert_progress(
|
||||
&h.app,
|
||||
&cookie,
|
||||
json!({ "manga_id": manga_id.to_string(), "page": 1 }),
|
||||
)
|
||||
.await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/me/read-progress/{manga_id}"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["new_chapters_count"], 0);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_single_manga_returns_404_when_unread(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
|
||||
@@ -127,6 +127,46 @@ async fn list_filters_by_content_warning(pool: PgPool) {
|
||||
assert_eq!(ids, want);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn denormalized_warnings_track_chapter_deletion(pool: PgPool) {
|
||||
// The denormalized manga_content_warnings table must stay in sync when the
|
||||
// underlying pages disappear. Deleting the only chapter (which cascade-
|
||||
// deletes its pages and their page_content_warnings) must drop the manga
|
||||
// from the include filter.
|
||||
let h = common::harness(pool.clone());
|
||||
let gory = seed_manga(&pool, "Gory", &[&["gore"]]).await;
|
||||
|
||||
// Present in the denorm table and matched by the filter.
|
||||
let count: i64 =
|
||||
sqlx::query_scalar("SELECT count(*) FROM manga_content_warnings WHERE manga_id = $1")
|
||||
.bind(gory)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(count, 1, "warning denormalized on insert");
|
||||
assert_eq!(list_ids(&h.app, "cw_include=gore").await, vec![gory.to_string()]);
|
||||
|
||||
// Delete the chapter → cascade removes pages + page_content_warnings →
|
||||
// trigger recomputes the (now empty) set for this manga.
|
||||
sqlx::query("DELETE FROM chapters WHERE manga_id = $1")
|
||||
.bind(gory)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let count: i64 =
|
||||
sqlx::query_scalar("SELECT count(*) FROM manga_content_warnings WHERE manga_id = $1")
|
||||
.bind(gory)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(count, 0, "warning removed after chapter (and pages) deleted");
|
||||
assert!(
|
||||
list_ids(&h.app, "cw_include=gore").await.is_empty(),
|
||||
"manga no longer matches the include filter"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_rejects_unknown_warning(pool: PgPool) {
|
||||
let h = common::harness(pool.clone());
|
||||
|
||||
@@ -99,6 +99,66 @@ async fn list_returns_total_count_independent_of_pagination(pool: PgPool) {
|
||||
assert_eq!(body["page"]["total"], 3);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_total_is_computed_only_on_the_first_page(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
for title in ["One Piece", "Berserk", "Vinland Saga"] {
|
||||
seed(&h.app, &cookie, title).await;
|
||||
}
|
||||
|
||||
// Page 1 (offset 0): total is the full population.
|
||||
let body0 = common::body_json(
|
||||
h.app
|
||||
.clone()
|
||||
.oneshot(common::get("/api/v1/mangas?limit=2&offset=0"))
|
||||
.await
|
||||
.unwrap(),
|
||||
)
|
||||
.await;
|
||||
assert_eq!(body0["page"]["total"], 3);
|
||||
|
||||
// Page 2 (offset 2): total is omitted (null) — the correlated count is
|
||||
// not recomputed on every page. Items still paginate correctly.
|
||||
let body1 = common::body_json(
|
||||
h.app
|
||||
.oneshot(common::get("/api/v1/mangas?limit=2&offset=2"))
|
||||
.await
|
||||
.unwrap(),
|
||||
)
|
||||
.await;
|
||||
assert!(
|
||||
body1["page"]["total"].is_null(),
|
||||
"total should be null past the first page, got {}",
|
||||
body1["page"]["total"]
|
||||
);
|
||||
assert_eq!(body1["items"].as_array().unwrap().len(), 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn search_treats_like_wildcards_literally(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
// Two long titles differing only at one position. `_` is a LIKE single-char
|
||||
// wildcard: unescaped, `%a_b%` matches BOTH ("axb" and "a_b"); escaped, only
|
||||
// the literal "a_b" title matches. The titles are long and the term short so
|
||||
// trigram similarity stays under threshold — the ILIKE branch decides.
|
||||
seed(&h.app, &cookie, "The Quick Brown Fox Jumps axb Over The Lazy Dog").await;
|
||||
seed(&h.app, &cookie, "The Quick Brown Fox Jumps a_b Over The Lazy Dog").await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get("/api/v1/mangas?search=a_b"))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(
|
||||
title_list(&body),
|
||||
vec!["The Quick Brown Fox Jumps a_b Over The Lazy Dog"],
|
||||
"the `_` in the search term must match literally, not as a wildcard"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn search_via_trigram_tolerates_typos(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
@@ -446,7 +506,7 @@ async fn list_tie_break_orders_equal_keys_by_ascending_id(pool: PgPool) {
|
||||
|
||||
// Paginating one row at a time reproduces that exact sequence — no overlap,
|
||||
// no gap — which only holds because the id tie-break makes the order total.
|
||||
let paged: Vec<String> = vec![page(1, 0).await, page(1, 1).await, page(1, 2).await]
|
||||
let paged: Vec<String> = [page(1, 0).await, page(1, 1).await, page(1, 2).await]
|
||||
.iter()
|
||||
.map(|b| b["items"][0]["title"].as_str().unwrap().to_string())
|
||||
.collect();
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
mod common;
|
||||
|
||||
use axum::http::StatusCode;
|
||||
use http_body_util::BodyExt;
|
||||
use serde_json::{json, Value};
|
||||
use sqlx::PgPool;
|
||||
use tower::ServiceExt;
|
||||
@@ -45,6 +46,65 @@ fn cover_form(bytes: &[u8]) -> MultipartBuilder {
|
||||
MultipartBuilder::new().add_file("cover", "cover.bin", "application/octet-stream", bytes)
|
||||
}
|
||||
|
||||
/// A real, decodable PNG (unlike `fake_png_bytes`, which is only magic bytes)
|
||||
/// so the thumbnail endpoint has something to resize.
|
||||
fn real_png(width: u32, height: u32) -> Vec<u8> {
|
||||
let mut buf = std::io::Cursor::new(Vec::new());
|
||||
image::RgbImage::from_pixel(width, height, image::Rgb([10, 120, 200]))
|
||||
.write_to(&mut buf, image::ImageFormat::Png)
|
||||
.unwrap();
|
||||
buf.into_inner()
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn files_serves_downscaled_thumbnail_variant(pool: PgPool) {
|
||||
let h = harness(pool);
|
||||
let (_, cookie) = register_user(&h.app).await;
|
||||
let manga = create_manga_with_cover(&h.app, &cookie, "Thumb", None).await;
|
||||
let id = id_of(&manga);
|
||||
|
||||
// Upload a real 800x400 PNG cover.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(put_multipart_with_cookie(
|
||||
&format!("/api/v1/mangas/{id}/cover"),
|
||||
cover_form(&real_png(800, 400)),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let key = format!("mangas/{id}/cover.png");
|
||||
|
||||
// ?w=320 serves a 320px-wide variant with the source content-type.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(get(&format!("/api/v1/files/{key}?w=320")))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
assert_eq!(
|
||||
resp.headers().get("content-type").unwrap().to_str().unwrap(),
|
||||
"image/png"
|
||||
);
|
||||
let body = resp.into_body().collect().await.unwrap().to_bytes();
|
||||
let thumb = image::load_from_memory(&body).unwrap();
|
||||
assert_eq!(thumb.width(), 320, "served a 320px-wide variant");
|
||||
assert_eq!(thumb.height(), 160, "aspect ratio preserved");
|
||||
|
||||
// The un-parametrised request still serves the full-resolution original.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(get(&format!("/api/v1/files/{key}")))
|
||||
.await
|
||||
.unwrap();
|
||||
let full = resp.into_body().collect().await.unwrap().to_bytes();
|
||||
assert_eq!(image::load_from_memory(&full).unwrap().width(), 800);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn put_cover_sets_path_when_none_existed(pool: PgPool) {
|
||||
let h = harness(pool);
|
||||
|
||||
@@ -326,6 +326,54 @@ async fn patch_updates_status_authors_and_genres(pool: PgPool) {
|
||||
assert_eq!(body["genres"][0]["name"], "Drama");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn patch_authors_updates_precomputed_sort_author(pool: PgPool) {
|
||||
// The precomputed mangas.sort_author (used by ?sort=author) must track
|
||||
// author edits: changing the alphabetically-first author reorders the
|
||||
// author sort. Two mangas whose relative author order flips after a PATCH.
|
||||
let h = common::harness(pool.clone());
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
|
||||
let a = id_of(&create_manga(&h.app, &cookie, json!({ "title": "A", "authors": ["Zeta"] })).await);
|
||||
let _b = id_of(&create_manga(&h.app, &cookie, json!({ "title": "B", "authors": ["Mid"] })).await);
|
||||
|
||||
// Initially: Mid < Zeta, so B before A.
|
||||
let ids = list_titles(&h.app, "sort=author&order=asc").await;
|
||||
assert_eq!(ids, vec!["B", "A"]);
|
||||
|
||||
// Re-author A to "Alpha", which now sorts first.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::patch_json_with_cookie(
|
||||
&format!("/api/v1/mangas/{a}"),
|
||||
json!({ "authors": ["Alpha"] }),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
|
||||
// Now A (Alpha) sorts before B (Mid) — proving sort_author was recomputed.
|
||||
let ids = list_titles(&h.app, "sort=author&order=asc").await;
|
||||
assert_eq!(ids, vec!["A", "B"]);
|
||||
}
|
||||
|
||||
async fn list_titles(app: &axum::Router, query: &str) -> Vec<String> {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::get(&format!("/api/v1/mangas?{query}")))
|
||||
.await
|
||||
.unwrap();
|
||||
let body = common::body_json(resp).await;
|
||||
body["items"]
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|m| m["title"].as_str().unwrap().to_string())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn patch_404_on_unknown_id(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
@@ -879,3 +927,25 @@ async fn bearer_authed_admin_cannot_edit_null_uploader(pool: PgPool) {
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn create_rejects_oversized_metadata_part(pool: PgPool) {
|
||||
// The metadata JSON part is capped well below the 200 MiB request limit so a
|
||||
// client can't force the server to buffer a huge JSON blob in memory before
|
||||
// any validation runs. A ~200 KiB description blows the metadata cap.
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
|
||||
let huge = "A".repeat(200 * 1024);
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::post_multipart_with_cookie(
|
||||
"/api/v1/mangas",
|
||||
MultipartBuilder::new()
|
||||
.add_json("metadata", json!({ "title": "Big", "description": huge })),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::PAYLOAD_TOO_LARGE);
|
||||
}
|
||||
|
||||
@@ -680,24 +680,123 @@ async fn aggregate_rejects_invalid_order(pool: PgPool) {
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn aggregate_with_text_param_is_501_with_stable_code(pool: PgPool) {
|
||||
// OCR text search isn't built yet; the param is accepted so adding
|
||||
// OCR won't break the wire shape, but rejected with a distinct
|
||||
// status + code. The code is the wire contract — clients pin on
|
||||
// `text_search_not_yet_supported`, not the message.
|
||||
let h = common::harness(pool);
|
||||
async fn aggregate_with_text_param_filters_by_ocr(pool: PgPool) {
|
||||
// OCR text search: a page tagged `funny` whose OCR contains "guts" is
|
||||
// returned for `&text=guts` on both aggregation endpoints, and excluded
|
||||
// for a query its OCR doesn't contain. The filter runs against the same
|
||||
// precomputed `search_doc` the OCR worker writes via `persist_analysis`.
|
||||
let h = common::harness(pool.clone());
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "M").await;
|
||||
let (_, page_id) = seed_chapter_with_page(&h.app, &cookie, manga_id, 1).await;
|
||||
assert_eq!(add_tag(&h.app, &cookie, &page_id, "funny").await, StatusCode::CREATED);
|
||||
|
||||
// Persist OCR text exactly as the ocrs backend would (via the same mapper).
|
||||
let page_uuid = Uuid::parse_str(&page_id).unwrap();
|
||||
let analysis =
|
||||
mangalord::analysis::ocr::lines_to_analysis(vec!["spilling the guts here".to_string()]);
|
||||
mangalord::repo::page_analysis::persist_analysis(&pool, page_uuid, &analysis, "ocrs")
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
for endpoint in ["chapters", "mangas"] {
|
||||
// Matching text → the tagged+OCR'd row is returned.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/me/page-tags/{endpoint}?tag=funny&text=guts"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK, "{endpoint} matching");
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"].as_array().unwrap().len(), 1, "{endpoint} matching items");
|
||||
assert_eq!(body["page"]["total"], 1, "{endpoint} matching total");
|
||||
|
||||
// Non-matching text → excluded.
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/me/page-tags/{endpoint}?tag=funny&text=zzzznomatch"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK, "{endpoint} non-matching");
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"].as_array().unwrap().len(), 0, "{endpoint} non-matching items");
|
||||
assert_eq!(body["page"]["total"], 0, "{endpoint} non-matching total");
|
||||
}
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn aggregate_text_param_counts_only_matching_pages(pool: PgPool) {
|
||||
// A chapter with TWO tagged pages where only ONE page's OCR matches the
|
||||
// text. `match_count` / `total` must reflect the filtered count (1), not
|
||||
// the tag-only count (2), and the sample thumbnails must contain only the
|
||||
// matching page. This is what a single-page test can't distinguish — it
|
||||
// pins the JOIN/placeholder wiring and the sample-subquery filter.
|
||||
let h = common::harness(pool.clone());
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "M").await;
|
||||
let (_, page_ids) = seed_chapter_with_n_pages(&h.app, &cookie, manga_id, 1, 2).await;
|
||||
for pid in &page_ids {
|
||||
assert_eq!(add_tag(&h.app, &cookie, pid, "funny").await, StatusCode::CREATED);
|
||||
}
|
||||
|
||||
// Page 0 OCR contains "guts"; page 1 OCR contains only "filler".
|
||||
let p0 = Uuid::parse_str(&page_ids[0]).unwrap();
|
||||
let p1 = Uuid::parse_str(&page_ids[1]).unwrap();
|
||||
let a0 = mangalord::analysis::ocr::lines_to_analysis(vec!["the guts spill out".to_string()]);
|
||||
let a1 = mangalord::analysis::ocr::lines_to_analysis(vec!["just filler text".to_string()]);
|
||||
mangalord::repo::page_analysis::persist_analysis(&pool, p0, &a0, "ocrs").await.unwrap();
|
||||
mangalord::repo::page_analysis::persist_analysis(&pool, p1, &a1, "ocrs").await.unwrap();
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
"/api/v1/me/page-tags/chapters?tag=funny&text=guts",
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::NOT_IMPLEMENTED);
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["error"]["code"], "text_search_not_yet_supported");
|
||||
let items = body["items"].as_array().unwrap();
|
||||
assert_eq!(items.len(), 1, "one chapter row");
|
||||
assert_eq!(body["page"]["total"], 1);
|
||||
// Filtered match_count is the matching-page count, not the tag count.
|
||||
assert_eq!(items[0]["match_count"], 1, "only one page's OCR matches");
|
||||
// Sample thumbnails reflect the filter: only the matching page's key.
|
||||
let samples = items[0]["sample_storage_keys"].as_array().unwrap();
|
||||
assert_eq!(samples.len(), 1, "thumbnails restricted to matching pages");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn aggregate_blank_text_param_is_tag_only(pool: PgPool) {
|
||||
// A blank `text=` must not filter — it falls back to tag-only aggregation
|
||||
// (the page has a tag but no analysis row at all).
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "M").await;
|
||||
let (_, page_id) = seed_chapter_with_page(&h.app, &cookie, manga_id, 1).await;
|
||||
assert_eq!(add_tag(&h.app, &cookie, &page_id, "funny").await, StatusCode::CREATED);
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.oneshot(common::get_with_cookie(
|
||||
"/api/v1/me/page-tags/chapters?tag=funny&text=",
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["items"].as_array().unwrap().len(), 1);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
|
||||
@@ -149,6 +149,56 @@ async fn private_mode_blocks_register_even_when_self_register_enabled(pool: PgPo
|
||||
assert_eq!(body["error"]["code"], "forbidden");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn private_mode_serves_files_as_private_not_public_cacheable(pool: PgPool) {
|
||||
// In private mode /files is auth-gated, so a served blob must NOT carry a
|
||||
// `public` Cache-Control: a shared cache / CDN would otherwise store it and
|
||||
// hand it to anonymous clients, defeating the gate. It should be `private`.
|
||||
//
|
||||
// Register via a public harness on the shared pool so the session exists,
|
||||
// then upload + fetch a cover through the private harness (same storage).
|
||||
let public = common::harness(pool.clone());
|
||||
let (_, cookie) = common::register_user(&public.app).await;
|
||||
|
||||
let private = common::harness_with_private_mode(pool);
|
||||
let form = common::MultipartBuilder::new()
|
||||
.add_json("metadata", json!({ "title": "Secret Library" }))
|
||||
.add_file("cover", "cover.png", "image/png", &common::fake_png_bytes());
|
||||
let created = private
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_multipart_with_cookie(
|
||||
"/api/v1/mangas",
|
||||
form,
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(created.status(), StatusCode::CREATED);
|
||||
let body = common::body_json(created).await;
|
||||
let key = body["cover_image_path"].as_str().unwrap().to_string();
|
||||
|
||||
let resp = private
|
||||
.app
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/files/{key}"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let cc = resp
|
||||
.headers()
|
||||
.get(axum::http::header::CACHE_CONTROL)
|
||||
.unwrap()
|
||||
.to_str()
|
||||
.unwrap();
|
||||
assert!(
|
||||
cc.starts_with("private") && !cc.contains("public"),
|
||||
"private-mode blobs must be `private`, not `public`; got: {cc}"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn auth_config_reports_private_mode_and_effective_self_register(pool: PgPool) {
|
||||
let h = common::harness_with_private_mode(pool);
|
||||
|
||||
133
backend/tests/api_reactions.rs
Normal file
133
backend/tests/api_reactions.rs
Normal file
@@ -0,0 +1,133 @@
|
||||
mod common;
|
||||
|
||||
use axum::http::StatusCode;
|
||||
use serde_json::json;
|
||||
use sqlx::PgPool;
|
||||
use tower::ServiceExt;
|
||||
use uuid::Uuid;
|
||||
|
||||
async fn put_reaction(
|
||||
app: &axum::Router,
|
||||
cookie: &str,
|
||||
manga_id: Uuid,
|
||||
reaction: &str,
|
||||
) -> StatusCode {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::put_json_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/reaction"),
|
||||
json!({ "reaction": reaction }),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
resp.status()
|
||||
}
|
||||
|
||||
async fn get_reaction(app: &axum::Router, cookie: &str, manga_id: Uuid) -> serde_json::Value {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie(
|
||||
&format!("/api/v1/me/reactions/{manga_id}"),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
common::body_json(resp).await
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn put_creates_and_reads_back(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
|
||||
assert_eq!(put_reaction(&h.app, &cookie, manga_id, "like").await, StatusCode::OK);
|
||||
let body = get_reaction(&h.app, &cookie, manga_id).await;
|
||||
assert_eq!(body["reaction"], "like");
|
||||
assert_eq!(body["manga_id"], manga_id.to_string());
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn put_toggles_like_to_dislike(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
|
||||
let _ = put_reaction(&h.app, &cookie, manga_id, "like").await;
|
||||
assert_eq!(put_reaction(&h.app, &cookie, manga_id, "dislike").await, StatusCode::OK);
|
||||
assert_eq!(get_reaction(&h.app, &cookie, manga_id).await["reaction"], "dislike");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn delete_clears_the_reaction(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
let _ = put_reaction(&h.app, &cookie, manga_id, "like").await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::delete_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/reaction"),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::NO_CONTENT);
|
||||
assert_eq!(get_reaction(&h.app, &cookie, manga_id).await["reaction"], serde_json::Value::Null);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_unset_returns_null(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
assert_eq!(get_reaction(&h.app, &cookie, manga_id).await["reaction"], serde_json::Value::Null);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reactions_are_per_user(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, a) = common::register_user(&h.app).await;
|
||||
let (_, b) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &a, "Berserk").await;
|
||||
let _ = put_reaction(&h.app, &a, manga_id, "like").await;
|
||||
// B sees no reaction of their own.
|
||||
assert_eq!(get_reaction(&h.app, &b, manga_id).await["reaction"], serde_json::Value::Null);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn put_unknown_manga_is_404(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
assert_eq!(
|
||||
put_reaction(&h.app, &cookie, Uuid::new_v4(), "like").await,
|
||||
StatusCode::NOT_FOUND
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn put_invalid_reaction_is_422(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
assert_eq!(
|
||||
put_reaction(&h.app, &cookie, manga_id, "meh").await,
|
||||
StatusCode::UNPROCESSABLE_ENTITY
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn get_requires_authentication(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get(&format!("/api/v1/me/reactions/{}", Uuid::new_v4())))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::UNAUTHORIZED);
|
||||
}
|
||||
182
backend/tests/api_recommendations.rs
Normal file
182
backend/tests/api_recommendations.rs
Normal file
@@ -0,0 +1,182 @@
|
||||
mod common;
|
||||
|
||||
use axum::http::StatusCode;
|
||||
use serde_json::json;
|
||||
use sqlx::PgPool;
|
||||
use tower::ServiceExt;
|
||||
use uuid::Uuid;
|
||||
|
||||
async fn attach_tag(app: &axum::Router, cookie: &str, manga_id: Uuid, name: &str) {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/tags"),
|
||||
json!({ "name": name }),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(resp.status().is_success(), "attach_tag: {}", resp.status());
|
||||
}
|
||||
|
||||
async fn set_reaction(app: &axum::Router, cookie: &str, manga_id: Uuid, reaction: &str) {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::put_json_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/reaction"),
|
||||
json!({ "reaction": reaction }),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
}
|
||||
|
||||
async fn bookmark(app: &axum::Router, cookie: &str, manga_id: Uuid) {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::post_json_with_cookie(
|
||||
"/api/v1/bookmarks",
|
||||
json!({ "manga_id": manga_id.to_string() }),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::CREATED);
|
||||
}
|
||||
|
||||
async fn mark_read(app: &axum::Router, cookie: &str, manga_id: Uuid) {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::put_json_with_cookie(
|
||||
"/api/v1/me/read-progress",
|
||||
json!({ "manga_id": manga_id.to_string(), "page": 1 }),
|
||||
cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
}
|
||||
|
||||
/// Recommended manga titles, in ranked order.
|
||||
async fn recommend(app: &axum::Router, cookie: &str) -> Vec<String> {
|
||||
let resp = app
|
||||
.clone()
|
||||
.oneshot(common::get_with_cookie("/api/v1/me/recommendations", cookie))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::OK);
|
||||
let body = common::body_json(resp).await;
|
||||
body["items"]
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|m| m["title"].as_str().unwrap().to_string())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn likes_drive_recommendations(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let liked = common::seed_manga_via_api(&h.app, &cookie, "Liked").await;
|
||||
let similar = common::seed_manga_via_api(&h.app, &cookie, "Similar").await;
|
||||
let unrelated = common::seed_manga_via_api(&h.app, &cookie, "Unrelated").await;
|
||||
attach_tag(&h.app, &cookie, liked, "action").await;
|
||||
attach_tag(&h.app, &cookie, similar, "action").await;
|
||||
attach_tag(&h.app, &cookie, unrelated, "sports").await;
|
||||
|
||||
set_reaction(&h.app, &cookie, liked, "like").await;
|
||||
|
||||
let recs = recommend(&h.app, &cookie).await;
|
||||
assert!(recs.contains(&"Similar".to_string()), "recs: {recs:?}");
|
||||
assert!(!recs.contains(&"Unrelated".to_string()), "recs: {recs:?}");
|
||||
// The liked manga itself is not recommended back.
|
||||
assert!(!recs.contains(&"Liked".to_string()), "recs: {recs:?}");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dislike_downranks_shared_tags(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let liked = common::seed_manga_via_api(&h.app, &cookie, "Liked").await;
|
||||
let disliked = common::seed_manga_via_api(&h.app, &cookie, "Disliked").await;
|
||||
let good = common::seed_manga_via_api(&h.app, &cookie, "Good").await;
|
||||
let bad = common::seed_manga_via_api(&h.app, &cookie, "Bad").await;
|
||||
attach_tag(&h.app, &cookie, liked, "action").await;
|
||||
attach_tag(&h.app, &cookie, good, "action").await; // shares the liked tag
|
||||
attach_tag(&h.app, &cookie, disliked, "gore").await;
|
||||
attach_tag(&h.app, &cookie, bad, "gore").await; // shares the disliked tag
|
||||
|
||||
set_reaction(&h.app, &cookie, liked, "like").await;
|
||||
set_reaction(&h.app, &cookie, disliked, "dislike").await;
|
||||
|
||||
let recs = recommend(&h.app, &cookie).await;
|
||||
assert!(recs.contains(&"Good".to_string()), "recs: {recs:?}");
|
||||
// Net-negative (disliked-tag) candidate is dropped.
|
||||
assert!(!recs.contains(&"Bad".to_string()), "recs: {recs:?}");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn bookmark_counts_as_half_a_like(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let liked = common::seed_manga_via_api(&h.app, &cookie, "Liked").await;
|
||||
let booked = common::seed_manga_via_api(&h.app, &cookie, "Booked").await;
|
||||
let from_like = common::seed_manga_via_api(&h.app, &cookie, "FromLike").await;
|
||||
let from_bookmark = common::seed_manga_via_api(&h.app, &cookie, "FromBookmark").await;
|
||||
attach_tag(&h.app, &cookie, liked, "tliked").await;
|
||||
attach_tag(&h.app, &cookie, from_like, "tliked").await; // affinity 1.0
|
||||
attach_tag(&h.app, &cookie, booked, "tbooked").await;
|
||||
attach_tag(&h.app, &cookie, from_bookmark, "tbooked").await; // affinity 0.5
|
||||
|
||||
set_reaction(&h.app, &cookie, liked, "like").await;
|
||||
bookmark(&h.app, &cookie, booked).await;
|
||||
|
||||
let recs = recommend(&h.app, &cookie).await;
|
||||
// Both recommended, but the like-derived one outranks the bookmark-derived.
|
||||
let i_like = recs.iter().position(|t| t == "FromLike");
|
||||
let i_book = recs.iter().position(|t| t == "FromBookmark");
|
||||
assert!(i_like.is_some() && i_book.is_some(), "recs: {recs:?}");
|
||||
assert!(i_like < i_book, "like should outrank bookmark; recs: {recs:?}");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn excludes_already_seen(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let liked = common::seed_manga_via_api(&h.app, &cookie, "Liked").await;
|
||||
let fresh = common::seed_manga_via_api(&h.app, &cookie, "Fresh").await;
|
||||
let already_read = common::seed_manga_via_api(&h.app, &cookie, "AlreadyRead").await;
|
||||
for m in [liked, fresh, already_read] {
|
||||
attach_tag(&h.app, &cookie, m, "action").await;
|
||||
}
|
||||
set_reaction(&h.app, &cookie, liked, "like").await;
|
||||
mark_read(&h.app, &cookie, already_read).await;
|
||||
|
||||
let recs = recommend(&h.app, &cookie).await;
|
||||
assert!(recs.contains(&"Fresh".to_string()), "recs: {recs:?}");
|
||||
assert!(!recs.contains(&"AlreadyRead".to_string()), "recs: {recs:?}");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn empty_without_signals(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let m = common::seed_manga_via_api(&h.app, &cookie, "Whatever").await;
|
||||
attach_tag(&h.app, &cookie, m, "action").await;
|
||||
|
||||
assert!(recommend(&h.app, &cookie).await.is_empty());
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn requires_authentication(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::get("/api/v1/me/recommendations"))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::UNAUTHORIZED);
|
||||
}
|
||||
@@ -279,7 +279,7 @@ async fn tag_autocomplete_returns_matches_ordered_by_similarity(pool: PgPool) {
|
||||
.iter()
|
||||
.map(|t| t["name"].as_str().unwrap())
|
||||
.collect();
|
||||
assert!(names.iter().any(|n| *n == "Mystery"));
|
||||
assert!(names.iter().any(|n| *n == "Murder Mystery"));
|
||||
assert!(!names.iter().any(|n| *n == "Comedy"));
|
||||
assert!(names.contains(&"Mystery"));
|
||||
assert!(names.contains(&"Murder Mystery"));
|
||||
assert!(!names.contains(&"Comedy"));
|
||||
}
|
||||
|
||||
@@ -172,6 +172,13 @@ async fn files_endpoint_streams_in_multiple_frames(pool: PgPool) {
|
||||
resp.headers().get("x-content-type-options").unwrap(),
|
||||
"nosniff"
|
||||
);
|
||||
// Blobs are content-addressed by unguessable, immutable keys, so they're
|
||||
// safe to cache forever — this is what makes the reader's page + next-
|
||||
// chapter preloading actually hit cache instead of re-downloading.
|
||||
assert_eq!(
|
||||
resp.headers().get(header::CACHE_CONTROL).unwrap(),
|
||||
"public, max-age=31536000, immutable"
|
||||
);
|
||||
|
||||
let mut body = resp.into_body();
|
||||
let mut frames = 0usize;
|
||||
@@ -328,6 +335,42 @@ async fn create_chapter_rejects_when_no_pages_with_422(pool: PgPool) {
|
||||
assert!(body["error"]["details"]["page"].is_string());
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn create_chapter_rejects_over_page_cap_with_413(pool: PgPool) {
|
||||
// Cap of 2 pages: a 3-page upload is refused once the third `page` part
|
||||
// arrives, before any chapter row is written.
|
||||
let h = common::harness_with_page_cap(pool.clone(), 2);
|
||||
let (_, cookie) = common::register_user(&h.app).await;
|
||||
let manga_id = common::seed_manga_via_api(&h.app, &cookie, "Berserk").await;
|
||||
|
||||
let resp = h
|
||||
.app
|
||||
.clone()
|
||||
.oneshot(common::post_multipart_with_cookie(
|
||||
&format!("/api/v1/mangas/{manga_id}/chapters"),
|
||||
MultipartBuilder::new()
|
||||
.add_json("metadata", json!({ "number": 1 }))
|
||||
.add_file("page", "1.png", "image/png", &common::fake_png_bytes())
|
||||
.add_file("page", "2.png", "image/png", &common::fake_png_bytes())
|
||||
.add_file("page", "3.png", "image/png", &common::fake_png_bytes()),
|
||||
&cookie,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(resp.status(), StatusCode::PAYLOAD_TOO_LARGE);
|
||||
let body = common::body_json(resp).await;
|
||||
assert_eq!(body["error"]["code"], "payload_too_large");
|
||||
|
||||
// Nothing persisted — the cap trips before the chapter transaction.
|
||||
let (chapter_count,): (i64,) =
|
||||
sqlx::query_as("SELECT count(*) FROM chapters WHERE manga_id = $1")
|
||||
.bind(manga_id)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(chapter_count, 0, "over-cap upload must not create a chapter");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn create_chapter_rejects_renamed_non_image_page(pool: PgPool) {
|
||||
let h = common::harness(pool);
|
||||
|
||||
@@ -79,6 +79,7 @@ fn harness_with_auth_config(
|
||||
// exercise without producing tens of MBs of bytes.
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter: 2000,
|
||||
},
|
||||
auth_limiter,
|
||||
// Default harness has no crawler daemon wired up; admin resync
|
||||
@@ -147,6 +148,27 @@ pub fn harness_with_auth_rate_limit(
|
||||
harness_with_auth_config(pool, storage, storage_dir, auth)
|
||||
}
|
||||
|
||||
/// Like [`harness_with_auth_rate_limit`] but also sets whether the backend
|
||||
/// trusts `X-Forwarded-For` (`AUTH_TRUSTED_PROXY`). Used to prove the per-IP
|
||||
/// rate-limit gate: with `trusted_proxy` on, distinct XFF hops get independent
|
||||
/// buckets; with it off, XFF is ignored and everything shares the global bucket.
|
||||
pub fn harness_with_auth_rate_limit_proxy(
|
||||
pool: PgPool,
|
||||
per_sec: u32,
|
||||
burst: u32,
|
||||
trusted_proxy: bool,
|
||||
) -> Harness {
|
||||
let storage_dir = tempfile::tempdir().expect("tempdir");
|
||||
let storage = Arc::new(LocalStorage::new(storage_dir.path()));
|
||||
let auth = AuthConfig {
|
||||
cookie_secure: false,
|
||||
trusted_proxy,
|
||||
rate_limit: mangalord::auth::rate_limit::RateLimitConfig { per_sec, burst },
|
||||
..AuthConfig::default()
|
||||
};
|
||||
harness_with_auth_config(pool, storage, storage_dir, auth)
|
||||
}
|
||||
|
||||
/// Like [`harness`] but slots a caller-supplied [`ResyncService`] stub
|
||||
/// into `AppState.resync`. Used by the admin resync tests so the
|
||||
/// endpoint path is exercised without standing up a real Chromium.
|
||||
@@ -170,6 +192,7 @@ pub fn harness_with_resync(
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter: 2000,
|
||||
},
|
||||
auth_limiter,
|
||||
runtime,
|
||||
@@ -203,6 +226,7 @@ pub fn harness_with_analysis(pool: PgPool) -> Harness {
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter: 2000,
|
||||
},
|
||||
auth_limiter,
|
||||
runtime: Arc::new(RuntimeControls::new(true)),
|
||||
@@ -218,6 +242,39 @@ pub fn harness_with_analysis(pool: PgPool) -> Harness {
|
||||
}
|
||||
}
|
||||
|
||||
/// Like [`harness`] but with a low `max_pages_per_chapter` so the chapter
|
||||
/// upload page-count cap is cheap to exercise.
|
||||
pub fn harness_with_page_cap(pool: PgPool, max_pages_per_chapter: usize) -> Harness {
|
||||
let storage_dir = tempfile::tempdir().expect("tempdir");
|
||||
let storage = Arc::new(LocalStorage::new(storage_dir.path()));
|
||||
let auth = AuthConfig {
|
||||
cookie_secure: false,
|
||||
..AuthConfig::default()
|
||||
};
|
||||
let auth_limiter = Arc::new(AuthRateLimiter::new(auth.rate_limit));
|
||||
let state = AppState {
|
||||
db: pool,
|
||||
storage,
|
||||
auth,
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter,
|
||||
},
|
||||
auth_limiter,
|
||||
runtime: Arc::new(RuntimeControls::new(false)),
|
||||
reloader: None,
|
||||
crawler_base: CrawlerConfig::default(),
|
||||
analysis_base: AnalysisConfig::default(),
|
||||
admin_allowed_origins: Arc::new(vec![TEST_ORIGIN.to_string()]),
|
||||
analysis_events: Arc::new(mangalord::analysis::events::AnalysisEvents::new()),
|
||||
};
|
||||
Harness {
|
||||
app: router(state),
|
||||
_storage_dir: storage_dir,
|
||||
}
|
||||
}
|
||||
|
||||
/// A [`DaemonReloader`] stub that records the configs it was asked to apply
|
||||
/// and flips the shared analysis gate, without spawning any real daemon. Lets
|
||||
/// settings tests assert that a `PUT` triggers a reload with the converted
|
||||
@@ -265,6 +322,7 @@ pub fn harness_with_settings_reloader(pool: PgPool) -> (Harness, Arc<StubReloade
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter: 2000,
|
||||
},
|
||||
auth_limiter,
|
||||
runtime,
|
||||
@@ -300,6 +358,7 @@ pub fn harness_with_admin_origins(pool: PgPool, origins: Vec<String>) -> Harness
|
||||
upload: UploadConfig {
|
||||
max_request_bytes: 4 * 1024 * 1024,
|
||||
max_file_bytes: 256 * 1024,
|
||||
max_pages_per_chapter: 2000,
|
||||
},
|
||||
auth_limiter,
|
||||
runtime: Arc::new(RuntimeControls::new(false)),
|
||||
@@ -376,6 +435,12 @@ impl Storage for FailingStorage {
|
||||
async fn size(&self, key: &str) -> Result<u64, StorageError> {
|
||||
self.inner.size(key).await
|
||||
}
|
||||
// Delegate straight to the inner filesystem rename — the fault
|
||||
// injection counts `put`/`put_stream` only, so promoting a staged page
|
||||
// to its final key never spuriously trips the injected failure.
|
||||
async fn rename(&self, from: &str, to: &str) -> Result<(), StorageError> {
|
||||
self.inner.rename(from, to).await
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn body_json(response: axum::response::Response) -> serde_json::Value {
|
||||
@@ -431,6 +496,20 @@ pub fn post_json_with_cookie(
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Like [`post_json_with_cookie`] but sends a raw (possibly malformed) body,
|
||||
/// for exercising body-parse error handling. Content-Type stays JSON and the
|
||||
/// CSRF Origin is attached so the request reaches the handler.
|
||||
pub fn post_raw_with_cookie(uri: &str, body: &str, cookie: &str) -> Request<Body> {
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(uri)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header(header::COOKIE, cookie)
|
||||
.header(header::ORIGIN, TEST_ORIGIN)
|
||||
.body(Body::from(body.to_string()))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Same as [`post_json_with_cookie`] but also attaches `Origin` (and
|
||||
/// optionally `Referer`) headers. Used by the admin CSRF tests to drive
|
||||
/// the cross-origin reject + allowed-origin accept paths.
|
||||
|
||||
@@ -62,6 +62,43 @@ async fn headless_browser_can_navigate_and_read_title() {
|
||||
handle.close().await.expect("close cleanly");
|
||||
}
|
||||
|
||||
/// Smoke-test the opt-in SSRF navigation guard (`CRAWLER_SSRF_INTERCEPT`).
|
||||
/// With interception ON, a normal (allowed) navigation must still complete —
|
||||
/// i.e. enabling CDP `Fetch` and installing the request handler must NOT wedge
|
||||
/// page loads. This is the regression the interceptor's fragility risks; it
|
||||
/// can only be exercised with a real Chromium, hence `#[ignore]`.
|
||||
///
|
||||
/// (The private-target *blocking* path is covered by the pure-logic unit tests
|
||||
/// in `crawler::intercept` — `verdict` / `is_blocked` — which don't need a
|
||||
/// browser.)
|
||||
#[tokio::test]
|
||||
#[ignore = "downloads Chromium; run with --ignored"]
|
||||
async fn ssrf_interception_does_not_wedge_allowed_navigation() {
|
||||
use mangalord::crawler::intercept;
|
||||
|
||||
const PAGE: &str =
|
||||
"data:text/html,<html><head><title>Guarded%20OK</title></head><body></body></html>";
|
||||
|
||||
intercept::set_enabled(true);
|
||||
let handle = browser::launch(LaunchOptions::headless())
|
||||
.await
|
||||
.expect("launch headless chromium");
|
||||
|
||||
// Route through the guarded opener (blank page -> Fetch.enable -> handler
|
||||
// -> goto). If the handler failed to continue the navigation, this would
|
||||
// hang until the test harness times out.
|
||||
let page = intercept::open_page(handle.browser(), PAGE)
|
||||
.await
|
||||
.expect("guarded open_page");
|
||||
page.wait_for_navigation().await.expect("wait for navigation");
|
||||
|
||||
let title = page.get_title().await.expect("get title");
|
||||
assert_eq!(title.as_deref(), Some("Guarded OK"));
|
||||
|
||||
handle.close().await.expect("close cleanly");
|
||||
intercept::set_enabled(false);
|
||||
}
|
||||
|
||||
/// Live end-to-end: navigate to a real page, get the rendered HTML, and
|
||||
/// parse it with `scraper`. ipify.org renders the visitor's public IP
|
||||
/// into the page DOM, so a successful run proves browser → render →
|
||||
|
||||
@@ -104,6 +104,20 @@ impl ChapterDispatcher for FailingDispatcher {
|
||||
}
|
||||
}
|
||||
|
||||
/// Always reports the browser unavailable — models a Chromium outage where
|
||||
/// `acquire()` fails. The job must be returned to `pending` WITHOUT burning an
|
||||
/// attempt, so an outage doesn't chew the backlog to `dead`.
|
||||
struct BrowserUnavailableDispatcher {
|
||||
seen: AtomicUsize,
|
||||
}
|
||||
#[async_trait::async_trait]
|
||||
impl ChapterDispatcher for BrowserUnavailableDispatcher {
|
||||
async fn dispatch(&self, _payload: JobPayload) -> anyhow::Result<SyncOutcome> {
|
||||
self.seen.fetch_add(1, Ordering::AcqRel);
|
||||
Ok(SyncOutcome::BrowserUnavailable)
|
||||
}
|
||||
}
|
||||
|
||||
/// Never completes — used to verify the worker's outer dispatch timeout.
|
||||
struct HangingDispatcher {
|
||||
seen: AtomicUsize,
|
||||
@@ -204,6 +218,58 @@ async fn shutdown_mid_dispatch_releases_lease_without_burning_attempt(pool: PgPo
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn browser_unavailable_releases_lease_without_burning_attempt(pool: PgPool) {
|
||||
// During a browser outage the dispatcher reports BrowserUnavailable. The
|
||||
// worker must return the job to `pending` with the attempt refunded
|
||||
// (attempts stays 0) instead of ack-failing it toward `dead`, so the whole
|
||||
// pending backlog survives the outage.
|
||||
enqueue_chapter_job(&pool).await;
|
||||
let dispatcher = Arc::new(BrowserUnavailableDispatcher {
|
||||
seen: AtomicUsize::new(0),
|
||||
});
|
||||
let session_expired = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let cancel = CancellationToken::new();
|
||||
let handle = daemon::spawn(
|
||||
pool.clone(),
|
||||
cancel.clone(),
|
||||
make_cfg(None, dispatcher.clone(), session_expired, 1),
|
||||
);
|
||||
|
||||
// Wait until the dispatcher has been invoked at least once (the job was
|
||||
// leased and deferred).
|
||||
let mut dispatched = false;
|
||||
for _ in 0..40 {
|
||||
if dispatcher.seen.load(Ordering::Acquire) >= 1 {
|
||||
dispatched = true;
|
||||
break;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(50)).await;
|
||||
}
|
||||
assert!(dispatched, "dispatcher must have been invoked");
|
||||
|
||||
handle.shutdown().await;
|
||||
|
||||
assert_eq!(
|
||||
count_state(&pool, "dead").await,
|
||||
0,
|
||||
"browser outage must not dead-letter the job"
|
||||
);
|
||||
assert_eq!(
|
||||
count_state(&pool, "pending").await,
|
||||
1,
|
||||
"job returns to pending after a browser-unavailable outcome"
|
||||
);
|
||||
let attempts: i32 = sqlx::query_scalar("SELECT attempts FROM crawler_jobs")
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
attempts, 0,
|
||||
"browser-unavailable must refund the lease attempt (no burn)"
|
||||
);
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn workers_drain_jobs_through_dispatcher(pool: PgPool) {
|
||||
enqueue_chapter_job(&pool).await;
|
||||
|
||||
@@ -718,42 +718,49 @@ async fn release_returns_to_pending_and_undoes_attempt_increment(pool: PgPool) {
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reap_done_deletes_old_rows_keeps_fresh(pool: PgPool) {
|
||||
// Two done rows: one old (updated_at 10 days ago), one fresh.
|
||||
let old_id = match jobs::enqueue(&pool, &chapter_content_payload(Uuid::new_v4()))
|
||||
.await
|
||||
.unwrap()
|
||||
{
|
||||
EnqueueResult::Inserted(id) => id,
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let fresh_id = match jobs::enqueue(&pool, &chapter_content_payload(Uuid::new_v4()))
|
||||
.await
|
||||
.unwrap()
|
||||
{
|
||||
EnqueueResult::Inserted(id) => id,
|
||||
_ => unreachable!(),
|
||||
};
|
||||
async fn reap_terminal_deletes_old_done_and_dead_keeps_fresh_and_active(pool: PgPool) {
|
||||
// Helper: enqueue a fresh pending job and return its id.
|
||||
async fn enqueue_one(pool: &PgPool) -> Uuid {
|
||||
match jobs::enqueue(pool, &chapter_content_payload(Uuid::new_v4()))
|
||||
.await
|
||||
.unwrap()
|
||||
{
|
||||
EnqueueResult::Inserted(id) => id,
|
||||
_ => unreachable!(),
|
||||
}
|
||||
}
|
||||
|
||||
let old_done = enqueue_one(&pool).await;
|
||||
let old_dead = enqueue_one(&pool).await;
|
||||
let fresh_done = enqueue_one(&pool).await;
|
||||
let fresh_dead = enqueue_one(&pool).await;
|
||||
let old_pending = enqueue_one(&pool).await;
|
||||
|
||||
// Old terminal rows (10 days) in both terminal states — both must reap.
|
||||
sqlx::query("UPDATE crawler_jobs SET state='done', updated_at = now() - interval '10 days' WHERE id = $1")
|
||||
.bind(old_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
.bind(old_done).execute(&pool).await.unwrap();
|
||||
sqlx::query("UPDATE crawler_jobs SET state='dead', updated_at = now() - interval '10 days' WHERE id = $1")
|
||||
.bind(old_dead).execute(&pool).await.unwrap();
|
||||
// Fresh terminal rows — inside the retention window, kept.
|
||||
sqlx::query("UPDATE crawler_jobs SET state='done' WHERE id = $1")
|
||||
.bind(fresh_id)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
.bind(fresh_done).execute(&pool).await.unwrap();
|
||||
sqlx::query("UPDATE crawler_jobs SET state='dead' WHERE id = $1")
|
||||
.bind(fresh_dead).execute(&pool).await.unwrap();
|
||||
// Old but still active (pending) — never reaped regardless of age.
|
||||
sqlx::query("UPDATE crawler_jobs SET updated_at = now() - interval '10 days' WHERE id = $1")
|
||||
.bind(old_pending).execute(&pool).await.unwrap();
|
||||
|
||||
let deleted = jobs::reap_done(&pool, 7).await.unwrap();
|
||||
assert_eq!(deleted, 1);
|
||||
let deleted = jobs::reap_terminal(&pool, 7).await.unwrap();
|
||||
assert_eq!(deleted, 2, "both old done and old dead rows are reaped");
|
||||
|
||||
let remaining: Vec<Uuid> = sqlx::query_scalar("SELECT id FROM crawler_jobs ORDER BY id")
|
||||
let mut remaining: Vec<Uuid> = sqlx::query_scalar("SELECT id FROM crawler_jobs")
|
||||
.fetch_all(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(remaining, vec![fresh_id], "only fresh row remains");
|
||||
remaining.sort();
|
||||
let mut expected = vec![fresh_done, fresh_dead, old_pending];
|
||||
expected.sort();
|
||||
assert_eq!(remaining, expected, "fresh terminal + active-pending rows survive");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
@@ -840,7 +847,7 @@ async fn lease_ties_on_scheduled_at_break_by_created_at(pool: PgPool) {
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn reap_done_zero_is_a_no_op(pool: PgPool) {
|
||||
async fn reap_terminal_zero_is_a_no_op(pool: PgPool) {
|
||||
let id = match jobs::enqueue(&pool, &chapter_content_payload(Uuid::new_v4()))
|
||||
.await
|
||||
.unwrap()
|
||||
@@ -854,7 +861,7 @@ async fn reap_done_zero_is_a_no_op(pool: PgPool) {
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let deleted = jobs::reap_done(&pool, 0).await.unwrap();
|
||||
let deleted = jobs::reap_terminal(&pool, 0).await.unwrap();
|
||||
assert_eq!(deleted, 0);
|
||||
assert_eq!(job_count(&pool).await, 1);
|
||||
}
|
||||
|
||||
@@ -674,6 +674,110 @@ async fn arbitrary_genres_from_source_get_inserted(pool: PgPool) {
|
||||
assert_eq!(webtoons_count.0, 1, "case-insensitive lookup reuses the existing row");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn genre_dedup_survives_a_manga_linked_to_two_variants(pool: PgPool) {
|
||||
// Regression for a migration-0038 collision: a manga linked to TWO
|
||||
// non-canonical case-variants of one genre. A repoint-UPDATE would set both
|
||||
// rows to the canonical id in one statement -> manga_genres PK violation ->
|
||||
// migration rollback -> boot failure. #[sqlx::test] migrates a clean DB, so
|
||||
// recreate the dirty pre-index state and replay 0038's healing statements;
|
||||
// they must complete and leave exactly the one canonical link. (Mirrors the
|
||||
// healing SQL in migrations/0038_genres_name_lower_unique.sql.)
|
||||
let manga = Uuid::new_v4();
|
||||
sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, 'T')")
|
||||
.bind(manga)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Drop the live guard so we can plant case-variant genres.
|
||||
sqlx::query("DROP INDEX genres_name_lower_uniq").execute(&pool).await.unwrap();
|
||||
let keep = Uuid::parse_str("00000000-0000-0000-0000-000000000001").unwrap();
|
||||
let dup2 = Uuid::parse_str("00000000-0000-0000-0000-000000000002").unwrap();
|
||||
let dup3 = Uuid::parse_str("00000000-0000-0000-0000-000000000003").unwrap();
|
||||
for (id, name) in [(keep, "Zzz"), (dup2, "zzz"), (dup3, "ZZZ")] {
|
||||
sqlx::query("INSERT INTO genres (id, name) VALUES ($1, $2)")
|
||||
.bind(id)
|
||||
.bind(name)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
}
|
||||
// The manga links to the two NON-canonical variants, not the canonical one.
|
||||
for gid in [dup2, dup3] {
|
||||
sqlx::query("INSERT INTO manga_genres (manga_id, genre_id) VALUES ($1, $2)")
|
||||
.bind(manga)
|
||||
.bind(gid)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Step 1 — collision-proof canonical-link backfill (the crux of the fix).
|
||||
sqlx::query(
|
||||
"INSERT INTO manga_genres (manga_id, genre_id) \
|
||||
SELECT DISTINCT mg.manga_id, k.keep_id \
|
||||
FROM manga_genres mg JOIN genres g ON mg.genre_id = g.id \
|
||||
JOIN (SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id \
|
||||
FROM genres GROUP BY lower(name)) k ON lower(g.name) = k.lname \
|
||||
WHERE g.id <> k.keep_id \
|
||||
ON CONFLICT (manga_id, genre_id) DO NOTHING",
|
||||
)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.expect("collision-proof INSERT must not raise a manga_genres PK violation");
|
||||
// Step 2 — drop the non-canonical links.
|
||||
sqlx::query(
|
||||
"DELETE FROM manga_genres mg USING genres g, \
|
||||
(SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id \
|
||||
FROM genres GROUP BY lower(name)) k \
|
||||
WHERE mg.genre_id = g.id AND lower(g.name) = k.lname AND g.id <> k.keep_id",
|
||||
)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
// Step 3 — remove orphaned duplicate genres.
|
||||
sqlx::query(
|
||||
"DELETE FROM genres g USING \
|
||||
(SELECT lower(name) AS lname, (array_agg(id ORDER BY id))[1] AS keep_id \
|
||||
FROM genres GROUP BY lower(name)) k \
|
||||
WHERE lower(g.name) = k.lname AND g.id <> k.keep_id",
|
||||
)
|
||||
.execute(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Exactly the canonical link remains, and the guard re-creates cleanly.
|
||||
let links: Vec<Uuid> =
|
||||
sqlx::query_scalar("SELECT genre_id FROM manga_genres WHERE manga_id = $1")
|
||||
.bind(manga)
|
||||
.fetch_all(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(links, vec![keep], "manga keeps exactly the canonical genre link");
|
||||
sqlx::query("CREATE UNIQUE INDEX genres_name_lower_uniq ON genres (lower(name))")
|
||||
.execute(&pool)
|
||||
.await
|
||||
.expect("no residual case-variant duplicates remain");
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn genres_reject_case_variant_duplicates_at_the_db(pool: PgPool) {
|
||||
// The sequential pre-check in sync_genres dedups the common case, but two
|
||||
// concurrent inserts could each miss it. The lower(name) unique index (0038)
|
||||
// is the race backstop: a case variant of an existing genre must be rejected
|
||||
// by the DB itself. "Action" is seeded by 0009.
|
||||
let dup = sqlx::query("INSERT INTO genres (name) VALUES ('action')")
|
||||
.execute(&pool)
|
||||
.await;
|
||||
let err = dup.expect_err("a case-variant of a seeded genre must violate the unique index");
|
||||
let msg = err.to_string();
|
||||
assert!(
|
||||
msg.contains("genres_name_lower_uniq") || msg.contains("unique") || msg.contains("duplicate"),
|
||||
"expected a unique-violation, got: {msg}"
|
||||
);
|
||||
}
|
||||
|
||||
/// User-attached tags (rows with non-NULL `added_by` in `manga_tags`)
|
||||
/// must survive a crawler upsert. The crawler owns source-attached tags
|
||||
/// (added_by IS NULL); user attachments are owned by the user who made
|
||||
@@ -1175,7 +1279,7 @@ async fn list_for_manga_returns_source_order_reversed(pool: PgPool) {
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn list_for_manga_places_null_source_index_last(pool: PgPool) {
|
||||
async fn list_for_manga_interleaves_null_source_index_by_number(pool: PgPool) {
|
||||
crawler::ensure_source(&pool, "target", "T", "https://x.example")
|
||||
.await
|
||||
.unwrap();
|
||||
@@ -1184,23 +1288,24 @@ async fn list_for_manga_places_null_source_index_last(pool: PgPool) {
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Crawled chapters get source_index 0 and 1; the upload path leaves
|
||||
// it NULL. NULLS LAST plus the (number, created_at) tail means the
|
||||
// upload sits after both crawled rows even though its number is in
|
||||
// the middle.
|
||||
// Crawled chapters, newest-first in the source DOM (so chapter 3 is at
|
||||
// source_index 0 and chapter 1 at source_index 1). Reversed for display that
|
||||
// is [Ch.1, Ch.3]. The uploaded chapter 2 must slot BETWEEN them by number,
|
||||
// not fall to the end — that was the bug (NULLS LAST dumped it last
|
||||
// regardless of its number).
|
||||
let crawled = vec![
|
||||
SourceChapterRef {
|
||||
source_chapter_key: "a".into(),
|
||||
number: 1,
|
||||
title: Some("Ch.1".into()),
|
||||
url: "https://x.example/foo/a".into(),
|
||||
},
|
||||
SourceChapterRef {
|
||||
source_chapter_key: "b".into(),
|
||||
number: 3,
|
||||
title: Some("Ch.3".into()),
|
||||
url: "https://x.example/foo/b".into(),
|
||||
},
|
||||
SourceChapterRef {
|
||||
source_chapter_key: "a".into(),
|
||||
number: 1,
|
||||
title: Some("Ch.1".into()),
|
||||
url: "https://x.example/foo/a".into(),
|
||||
},
|
||||
];
|
||||
crawler::sync_manga_chapters(&pool, "target", up.manga_id, &crawled)
|
||||
.await
|
||||
@@ -1220,11 +1325,11 @@ async fn list_for_manga_places_null_source_index_last(pool: PgPool) {
|
||||
assert_eq!(
|
||||
titles,
|
||||
vec![
|
||||
"Ch.3".to_string(),
|
||||
"Ch.1".to_string(),
|
||||
"User upload Ch.2".to_string(),
|
||||
"Ch.3".to_string(),
|
||||
],
|
||||
"crawled rows ordered by reversed source_index; user upload \
|
||||
(NULL source_index) falls through to the end",
|
||||
"uploaded chapter (NULL source_index) interleaves by number between the \
|
||||
crawled rows instead of falling to the end",
|
||||
);
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user