From d6ac648ac96086cf9074b16141a5cf0100d34153 Mon Sep 17 00:00:00 2001 From: MechaCat02 Date: Sat, 6 Jun 2026 18:49:56 +0200 Subject: [PATCH] feat(admin): crawler observability dashboard + reliability hardening (0.55.0) Admin-only crawler dashboard backed by an SSE live-status stream, coordinated browser restart, runtime PHPSESSID refresh, dead-letter requeue, and a batch of reliability fixes. Closes everything from the two-pass audit (10 commits' worth) and bumps 0.52.0 -> 0.55.0. Backend: - New /admin/crawler/* surface (cookie-auth, RequireAdmin) split into status / control / dead_jobs / backlog modules. SSE stream composes in-memory status with DB-derived queue counts, memoizes the counts for 1s and debounces watch pokes for 250ms (~10x QPS reduction per subscriber). One-shot GET /admin/crawler shares the same compose path. - POST /admin/crawler/run gated by manual_pass_lock try_lock_owned (409 Conflict on overlapping click); browser restart goes through the coordinated_restart gate (drain + relaunch + auto-clear of the sticky session_expired flag on Ok). - Runtime PHPSESSID refresh via SessionController (allow-list validation, never logged, audit row carries SHA-256 fingerprint). Storage layer is repo::crawler::runtime_session_{load,persist}. - Dead-letter requeue with four scopes (all/manga/chapter/job); scope=all requires confirm:true; DISTINCT ON dedup keeps the partial unique index from rejecting requeues for chapters with multiple dead rows. SQL is four &'static str constants per scope. - StatusHandle + ChapterGuard / CoverGuard RAII model survives panics; last-writer-wins on cover so concurrent dispatches don't clobber each other's slot. Pure functions (should_stop / should_mark_clean_exit / should_abort_pass) with named regression tests. - Reliability bundle: per-lease heartbeat, jitter on retries, per-job timeout, circuit breaker on consecutive failures, BrowserManager coordinated restart gate, request fingerprint changes. - Streaming page download: Storage::put_stream trait method, LocalStorage impl atomic via temp + fsync + UUID-suffixed rename. Pages stream through with peak memory ~one HTTP chunk + 64-byte sniff prefix instead of one full image per dispatch. - New partial indexes (migration 0022): mangas_missing_cover_idx and crawler_jobs_dead_idx, both ordered by updated_at DESC to match the dashboard's LIMIT/OFFSET reads. - Security hardening: admin_csrf_guard (Origin/Referer allowlist on /admin/* mutations, opt-in via ADMIN_ALLOWED_ORIGINS), admin_no_store_guard (Cache-Control: no-store on admin responses), audit rows carry per-scope target_id. Frontend: - /admin/crawler page decomposed into lib/components/crawler/ (11 components: ProgressBar, SearchBar, CrawlerHero, CrawlerControls, ActiveChaptersCard, ActiveJobsTable, MissingCoversTable, DeadJobsTable, RestartConfirmModal, RequeueAllConfirmModal, SessionModal). Page is 532 LOC of orchestration; each component 22-148 LOC. - EventSource lifecycle wired to visibilitychange / pagehide / pageshow (BFCache); after 5 consecutive errors probes the status endpoint so a 401 routes through the global on401Hook instead of infinite silent reconnects. - Backlog $effect refetches debounced 500ms with per-loader AbortControllers; refresh after a control action only runs when the SSE stream is dead. - Inline requeue button on /admin/mangas patches the affected row's sync_state locally (no full chapter-list refetch); proper aria-label. Requeue-all gets its own confirm modal; both confirm modals autofocus Cancel. - SvelteKit reverse proxy bypasses its 5-minute AbortController for Accept: text/event-stream; pure shouldBypassProxyTimeout helper covered by unit tests. Config / docs: - New env vars (.env.example): ADMIN_ALLOWED_ORIGINS, CRAWLER_JOB_TIMEOUT_SECS, CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES, CRAWLER_BROWSER_RESTART_THRESHOLD. Co-Authored-By: Claude Opus 4.7 (1M context) --- .env.example | 31 + backend/Cargo.lock | 2 +- backend/Cargo.toml | 2 +- .../0022_crawler_observability_indexes.sql | 19 + backend/src/api/admin/crawler/backlog.rs | 67 ++ backend/src/api/admin/crawler/control.rs | 206 ++++++ backend/src/api/admin/crawler/dead_jobs.rs | 121 ++++ backend/src/api/admin/crawler/mod.rs | 49 ++ backend/src/api/admin/crawler/status.rs | 245 +++++++ backend/src/api/admin/mod.rs | 2 + backend/src/app.rs | 279 +++++++- backend/src/bin/crawler.rs | 7 + backend/src/config.rs | 73 ++ backend/src/crawler/browser_manager.rs | 227 +++++- backend/src/crawler/content.rs | 349 ++++++++-- backend/src/crawler/daemon.rs | 96 ++- backend/src/crawler/jobs.rs | 107 ++- backend/src/crawler/mod.rs | 2 + backend/src/crawler/pipeline.rs | 113 ++- backend/src/crawler/resync.rs | 4 +- backend/src/crawler/safety.rs | 25 + backend/src/crawler/session_control.rs | 184 +++++ backend/src/crawler/status.rs | 456 ++++++++++++ backend/src/repo/chapter.rs | 8 +- backend/src/repo/crawler.rs | 426 +++++++++++- backend/src/storage/local.rs | 93 ++- backend/src/storage/mod.rs | 35 + backend/tests/api_admin_crawler.rs | 654 ++++++++++++++++++ backend/tests/api_admin_role.rs | 2 + backend/tests/common/mod.rs | 91 ++- backend/tests/crawler_daemon.rs | 48 ++ backend/tests/crawler_dead_jobs.rs | 304 ++++++++ backend/tests/crawler_jobs.rs | 62 ++ backend/tests/repo_chapter.rs | 4 +- frontend/package.json | 2 +- frontend/src/hooks.server.test.ts | 36 +- frontend/src/hooks.server.ts | 30 +- frontend/src/lib/api/admin.test.ts | 168 ++++- frontend/src/lib/api/admin.ts | 191 ++++- frontend/src/lib/api/client.ts | 9 + .../crawler/ActiveChaptersCard.svelte | 112 +++ .../components/crawler/ActiveJobsTable.svelte | 101 +++ .../components/crawler/CrawlerControls.svelte | 43 ++ .../lib/components/crawler/CrawlerHero.svelte | 148 ++++ .../components/crawler/DeadJobsTable.svelte | 132 ++++ .../crawler/MissingCoversTable.svelte | 85 +++ .../lib/components/crawler/ProgressBar.svelte | 35 + .../crawler/RequeueAllConfirmModal.svelte | 32 + .../crawler/RestartConfirmModal.svelte | 42 ++ .../lib/components/crawler/SearchBar.svelte | 22 + .../components/crawler/SessionModal.svelte | 49 ++ frontend/src/routes/admin/+layout.svelte | 1 + .../src/routes/admin/crawler/+page.svelte | 532 ++++++++++++++ frontend/src/routes/admin/mangas/+page.svelte | 50 ++ 54 files changed, 6076 insertions(+), 137 deletions(-) create mode 100644 backend/migrations/0022_crawler_observability_indexes.sql create mode 100644 backend/src/api/admin/crawler/backlog.rs create mode 100644 backend/src/api/admin/crawler/control.rs create mode 100644 backend/src/api/admin/crawler/dead_jobs.rs create mode 100644 backend/src/api/admin/crawler/mod.rs create mode 100644 backend/src/api/admin/crawler/status.rs create mode 100644 backend/src/crawler/session_control.rs create mode 100644 backend/src/crawler/status.rs create mode 100644 backend/tests/api_admin_crawler.rs create mode 100644 backend/tests/crawler_dead_jobs.rs create mode 100644 frontend/src/lib/components/crawler/ActiveChaptersCard.svelte create mode 100644 frontend/src/lib/components/crawler/ActiveJobsTable.svelte create mode 100644 frontend/src/lib/components/crawler/CrawlerControls.svelte create mode 100644 frontend/src/lib/components/crawler/CrawlerHero.svelte create mode 100644 frontend/src/lib/components/crawler/DeadJobsTable.svelte create mode 100644 frontend/src/lib/components/crawler/MissingCoversTable.svelte create mode 100644 frontend/src/lib/components/crawler/ProgressBar.svelte create mode 100644 frontend/src/lib/components/crawler/RequeueAllConfirmModal.svelte create mode 100644 frontend/src/lib/components/crawler/RestartConfirmModal.svelte create mode 100644 frontend/src/lib/components/crawler/SearchBar.svelte create mode 100644 frontend/src/lib/components/crawler/SessionModal.svelte create mode 100644 frontend/src/routes/admin/crawler/+page.svelte diff --git a/.env.example b/.env.example index b3f6da3..1fa1ecc 100644 --- a/.env.example +++ b/.env.example @@ -52,6 +52,23 @@ AUTH_RATE_BURST=10 # on different hosts. Example: https://app.example.com,https://app.example.de CORS_ALLOWED_ORIGINS= +# ----- Admin CSRF allowlist ----- +# Browser origins (scheme + host[:port]) permitted to POST to +# /api/v1/admin/* mutating endpoints. Defends the session-cookie- +# authenticated admin surface against SameSite=Lax form-POST CSRF. +# Same shape as CORS_ALLOWED_ORIGINS (comma-separated). Compare against +# the request's Origin header (falling back to Referer when absent); +# safe methods (GET/HEAD/OPTIONS) are not checked, and requests with +# neither Origin nor Referer (curl, server-to-server callers) are +# always allowed. +# +# Default is empty: CSRF check disabled (operator opt-out). For a +# browser-exposed deployment this should be set to the SvelteKit +# origin, e.g. https://app.example.com. For a same-origin +# docker-compose deploy where only one origin exists, set the same +# value the browser uses. +ADMIN_ALLOWED_ORIGINS= + # ----- Upload limits ----- # Per-request body cap. axum rejects oversized requests with 413 before # our handlers run. Default 200 MiB. @@ -78,6 +95,20 @@ CRAWLER_MAX_IMAGE_BYTES=33554432 # and the `bin/crawler` CLI). 0 means no cap — let the source walker run # to completion. Useful for capped test runs against a new source. CRAWLER_LIMIT=0 + +# ----- Crawler reliability knobs ----- +# Hard upper bound on a single chapter-content job dispatch (seconds). +# A job that exceeds the budget is acked failed (with exponential +# backoff) instead of wedging a worker. Default 600s. +CRAWLER_JOB_TIMEOUT_SECS=600 +# Consecutive metadata-pass `fetch_manga` failures that abort the pass +# (circuit breaker for a source outage). The pass does NOT mark a clean +# exit, so the next tick does a recovery sweep. Default 10. +CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES=10 +# Consecutive transient chapter failures (after TOR recircuit is +# exhausted) that trigger an automatic coordinated browser restart. +# Default 3. +CRAWLER_BROWSER_RESTART_THRESHOLD=3 # Path to a system Chromium binary. When set, the crawler skips the # bundled-fetcher download. Required on platforms without a usable # upstream Chromium build (notably Linux_arm64 / Raspberry Pi). On diff --git a/backend/Cargo.lock b/backend/Cargo.lock index 441ecc1..78f890b 100644 --- a/backend/Cargo.lock +++ b/backend/Cargo.lock @@ -1470,7 +1470,7 @@ checksum = "c41e0c4fef86961ac6d6f8a82609f55f31b05e4fce149ac5710e439df7619ba4" [[package]] name = "mangalord" -version = "0.52.0" +version = "0.55.0" dependencies = [ "anyhow", "argon2", diff --git a/backend/Cargo.toml b/backend/Cargo.toml index f62838d..b45b810 100644 --- a/backend/Cargo.toml +++ b/backend/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "mangalord" -version = "0.52.0" +version = "0.55.0" edition = "2021" default-run = "mangalord" diff --git a/backend/migrations/0022_crawler_observability_indexes.sql b/backend/migrations/0022_crawler_observability_indexes.sql new file mode 100644 index 0000000..b13ef35 --- /dev/null +++ b/backend/migrations/0022_crawler_observability_indexes.sql @@ -0,0 +1,19 @@ +-- Partial indexes that back the admin crawler dashboard hot reads: +-- * mangas with no cover — drives count_missing_covers / +-- list_missing_cover_mangas (the cover backlog the metadata pass drains). +-- * dead jobs — drives list_dead_jobs / requeue_dead_jobs. +-- Both filters are highly selective in steady state (the working sets are a +-- tiny fraction of the full tables), so the partials stay small and hot. +-- ORDER BY updated_at DESC matches the LIMIT/OFFSET page reads. +-- +-- Not CONCURRENTLY: sqlx::migrate! wraps each migration in a transaction; +-- CREATE INDEX CONCURRENTLY can't run inside one. Tables are small at our +-- scale (online deploy still safe). + +CREATE INDEX IF NOT EXISTS mangas_missing_cover_idx + ON mangas (updated_at DESC) + WHERE cover_image_path IS NULL; + +CREATE INDEX IF NOT EXISTS crawler_jobs_dead_idx + ON crawler_jobs (updated_at DESC) + WHERE state = 'dead'; diff --git a/backend/src/api/admin/crawler/backlog.rs b/backend/src/api/admin/crawler/backlog.rs new file mode 100644 index 0000000..e7efdd7 --- /dev/null +++ b/backend/src/api/admin/crawler/backlog.rs @@ -0,0 +1,67 @@ +//! GET /admin/crawler/active-jobs — paginated `pending|running` chapters +//! GET /admin/crawler/covers — paginated mangas missing a cover +//! +//! These are pure DB-derived reads that drive two of the three +//! backlog tables on the dashboard. The third (dead jobs) lives in +//! the [`super::dead_jobs`] module because it also exposes a write. + +use axum::extract::{Query, State}; +use axum::routing::get; +use axum::{Json, Router}; +use serde::Deserialize; + +use crate::app::AppState; +use crate::auth::extractor::RequireAdmin; +use crate::error::AppResult; +use crate::repo; +use crate::repo::crawler::{ActiveJob, MissingCoverRow}; + +use super::default_limit; + +pub(super) fn routes() -> Router { + Router::new() + .route("/admin/crawler/active-jobs", get(list_active_jobs)) + .route("/admin/crawler/covers", get(list_covers)) +} + +/// Pagination + title-search params shared by the backlog list endpoints. +#[derive(Debug, Deserialize, Default)] +struct ListParams { + #[serde(default)] + search: Option, + #[serde(default = "default_limit")] + limit: i64, + #[serde(default)] + offset: i64, +} + +async fn list_active_jobs( + State(state): State, + _admin: RequireAdmin, + Query(params): Query, +) -> AppResult>> { + let limit = params.limit.clamp(1, 200); + let offset = params.offset.max(0); + let search = params.search.filter(|s| !s.trim().is_empty()); + let (items, total) = + repo::crawler::list_active_jobs(&state.db, search.as_deref(), limit, offset).await?; + Ok(Json(crate::api::pagination::PagedResponse::with_total( + items, limit, offset, total, + ))) +} + +async fn list_covers( + State(state): State, + _admin: RequireAdmin, + Query(params): Query, +) -> AppResult>> { + let limit = params.limit.clamp(1, 200); + let offset = params.offset.max(0); + let search = params.search.filter(|s| !s.trim().is_empty()); + let (items, total) = + repo::crawler::list_missing_cover_mangas(&state.db, search.as_deref(), limit, offset) + .await?; + Ok(Json(crate::api::pagination::PagedResponse::with_total( + items, limit, offset, total, + ))) +} diff --git a/backend/src/api/admin/crawler/control.rs b/backend/src/api/admin/crawler/control.rs new file mode 100644 index 0000000..32eece2 --- /dev/null +++ b/backend/src/api/admin/crawler/control.rs @@ -0,0 +1,206 @@ +//! POST /admin/crawler/run — trigger an out-of-cycle metadata pass +//! POST /admin/crawler/browser/restart — coordinated restart +//! POST /admin/crawler/session — refresh PHPSESSID +//! POST /admin/crawler/session/clear-expired — clear sticky expired flag +//! +//! All four mutate live in-process state on `CrawlerControl` (browser +//! manager, session controller, manual-pass mutex). They each emit an +//! `admin_audit` row and `status.poke()` so SSE subscribers see the +//! change instantly without polling. + +use axum::extract::State; +use axum::routing::post; +use axum::{Json, Router}; +use serde::{Deserialize, Serialize}; +use serde_json::json; + +use crate::app::AppState; +use crate::auth::extractor::RequireAdmin; +use crate::error::{AppError, AppResult}; +use crate::repo; + +use super::require_crawler; + +pub(super) fn routes() -> Router { + Router::new() + .route("/admin/crawler/run", post(run_now)) + .route("/admin/crawler/browser/restart", post(restart_browser)) + .route("/admin/crawler/session", post(update_session)) + .route( + "/admin/crawler/session/clear-expired", + post(clear_session_expired), + ) +} + +#[derive(Debug, Serialize)] +struct RunResponse { + started: bool, +} + +async fn run_now( + State(state): State, + admin: RequireAdmin, +) -> AppResult> { + let c = require_crawler(&state)?; + let mp = c.metadata_pass.as_ref().ok_or_else(|| { + AppError::ServiceUnavailable("no source configured (CRAWLER_START_URL unset)".into()) + })?; + // Operator-click dedup. A pass holds the lock for its entire run + // (minutes); a second click while it's in flight returns 409 instead + // of fanning a second pass onto the single browser lease. The daily + // cron does NOT take this lock — cron is single-fire by definition, + // and its own contention with a manual pass already serialises + // through the browser lease + advisory lock at a lower layer. + let pass_guard = c + .manual_pass_lock + .clone() + .try_lock_owned() + .map_err(|_| AppError::Conflict("manual metadata pass already running".into()))?; + let mp = std::sync::Arc::clone(mp); + // Fire-and-forget: the pass can run for minutes; the dashboard + // streams progress over SSE. The guard moves into the task so the + // lock is released only when the pass finishes (or the task panics). + tokio::spawn(async move { + let _pass_guard = pass_guard; + if let Err(e) = mp.run().await { + tracing::warn!(error = ?e, "manual metadata pass failed"); + } + }); + repo::admin_audit::insert(&state.db, admin.0.id, "crawler_run", "crawler", None, json!({})) + .await?; + Ok(Json(RunResponse { started: true })) +} + +#[derive(Debug, Serialize)] +struct RestartResponse { + ok: bool, + error: Option, +} + +async fn restart_browser( + State(state): State, + admin: RequireAdmin, +) -> AppResult> { + let c = require_crawler(&state)?; + let result = c.browser_manager.coordinated_restart(c.drain_deadline).await; + // A successful coordinated_restart re-runs on_launch, which re-injects + // PHPSESSID and re-probes — i.e. the session is live. Drop the sticky + // `session_expired` flag so chapter workers stop idling without + // requiring a second click on "Clear expired". + if result.is_ok() { + c.session.clear_expired(); + } + // Push the post-restart browser phase to live subscribers immediately. + c.status.poke(); + repo::admin_audit::insert( + &state.db, + admin.0.id, + "crawler_browser_restart", + "crawler", + None, + json!({ "ok": result.is_ok() }), + ) + .await?; + Ok(Json(match result { + Ok(()) => RestartResponse { + ok: true, + error: None, + }, + Err(e) => RestartResponse { + ok: false, + error: Some(format!("{e:#}")), + }, + })) +} + +#[derive(Debug, Deserialize)] +struct UpdateSessionRequest { + phpsessid: String, +} + +#[derive(Debug, Serialize)] +struct UpdateSessionResponse { + /// Whether the post-update browser relaunch + session probe succeeded. + valid: bool, + error: Option, +} + +async fn update_session( + State(state): State, + admin: RequireAdmin, + Json(body): Json, +) -> AppResult> { + let c = require_crawler(&state)?; + // Fingerprint BEFORE move so the raw value never reaches tracing or + // the audit row. SHA256-prefix is opaque to anyone reading the audit + // log but deterministic enough to correlate two updates of the same + // session value. + let fingerprint = phpsessid_fingerprint(&body.phpsessid); + c.session + .update(&body.phpsessid) + .await + .map_err(|e| AppError::InvalidInput(format!("{e:#}")))?; + // Relaunch the browser so on_launch re-injects the new cookie and + // re-probes — the restart's success IS the session-validity signal. + let probe = c.browser_manager.coordinated_restart(c.drain_deadline).await; + // Session + browser state changed — push to live subscribers. + c.status.poke(); + repo::admin_audit::insert( + &state.db, + admin.0.id, + "crawler_session_update", + "crawler", + None, + json!({ "valid": probe.is_ok(), "phpsessid_fingerprint": fingerprint }), + ) + .await?; + Ok(Json(match probe { + Ok(()) => UpdateSessionResponse { + valid: true, + error: None, + }, + Err(e) => UpdateSessionResponse { + valid: false, + error: Some(format!("{e:#}")), + }, + })) +} + +#[derive(Debug, Serialize)] +struct ClearExpiredResponse { + cleared: bool, +} + +async fn clear_session_expired( + State(state): State, + admin: RequireAdmin, +) -> AppResult> { + let c = require_crawler(&state)?; + c.session.clear_expired(); + // session.expired flipped — push to live subscribers. + c.status.poke(); + repo::admin_audit::insert( + &state.db, + admin.0.id, + "crawler_session_clear_expired", + "crawler", + None, + json!({}), + ) + .await?; + Ok(Json(ClearExpiredResponse { cleared: true })) +} + +/// Opaque, short fingerprint of a PHPSESSID for the admin audit log. +/// The first 8 hex chars of SHA-256 — enough to correlate two updates +/// of the same value without revealing the raw cookie. Reading the audit +/// row does not give an operator anything that can re-construct the +/// session. +fn phpsessid_fingerprint(sid: &str) -> String { + use sha2::{Digest, Sha256}; + let mut h = Sha256::new(); + h.update(sid.as_bytes()); + let digest = h.finalize(); + let hex: String = digest.iter().take(4).map(|b| format!("{b:02x}")).collect(); + hex +} diff --git a/backend/src/api/admin/crawler/dead_jobs.rs b/backend/src/api/admin/crawler/dead_jobs.rs new file mode 100644 index 0000000..5951d63 --- /dev/null +++ b/backend/src/api/admin/crawler/dead_jobs.rs @@ -0,0 +1,121 @@ +//! GET /admin/crawler/dead-jobs — paginated dead-letter list +//! POST /admin/crawler/dead-jobs/requeue — flip dead jobs back to pending +//! +//! Requeue scopes: +//! - `all` (requires `confirm: true` so a careless click / CSRF bait +//! can't flip the whole pile in one shot) +//! - `manga` (all dead jobs whose chapter belongs to a manga) +//! - `chapter` (all dead jobs for a single chapter) +//! - `job` (a single dead row by id) +//! +//! Each requeue emits an `admin_audit` row with the relevant +//! `target_id` populated, so an operator review post-incident can pin +//! exactly which scope was acted on. + +use axum::extract::{Query, State}; +use axum::routing::{get, post}; +use axum::{Json, Router}; +use serde::{Deserialize, Serialize}; +use serde_json::json; +use uuid::Uuid; + +use crate::app::AppState; +use crate::auth::extractor::RequireAdmin; +use crate::error::{AppError, AppResult}; +use crate::repo; +use crate::repo::crawler::{DeadJob, RequeueScope}; + +use super::default_limit; + +pub(super) fn routes() -> Router { + Router::new() + .route("/admin/crawler/dead-jobs", get(list_dead_jobs)) + .route("/admin/crawler/dead-jobs/requeue", post(requeue_dead_jobs)) +} + +#[derive(Debug, Deserialize, Default)] +struct DeadJobsParams { + #[serde(default)] + search: Option, + #[serde(default = "default_limit")] + limit: i64, + #[serde(default)] + offset: i64, +} + +async fn list_dead_jobs( + State(state): State, + _admin: RequireAdmin, + Query(params): Query, +) -> AppResult>> { + let limit = params.limit.clamp(1, 200); + let offset = params.offset.max(0); + let search = params.search.filter(|s| !s.trim().is_empty()); + let (items, total) = + repo::crawler::list_dead_jobs(&state.db, search.as_deref(), limit, offset).await?; + Ok(Json(crate::api::pagination::PagedResponse::with_total( + items, limit, offset, total, + ))) +} + +#[derive(Debug, Deserialize)] +#[serde(tag = "scope", rename_all = "snake_case")] +enum RequeueRequest { + /// `confirm: true` is required so a careless click / CSRF bait can't + /// flip the entire dead pile in one shot. Narrow scopes don't need it. + All { + #[serde(default)] + confirm: bool, + }, + Manga { manga_id: Uuid }, + Chapter { chapter_id: Uuid }, + Job { job_id: Uuid }, +} + +#[derive(Debug, Serialize)] +struct RequeueResponse { + requeued: u64, +} + +async fn requeue_dead_jobs( + State(state): State, + admin: RequireAdmin, + Json(body): Json, +) -> AppResult> { + // Reject scope=all without an explicit confirm so a single click or + // CSRF bait can't flip the whole dead pile. Narrow scopes don't need + // the confirmation — the operator already named a specific target. + if let RequeueRequest::All { confirm: false } = &body { + return Err(AppError::InvalidInput( + "confirm: true is required for scope=all".into(), + )); + } + let (scope, target_kind, target_id) = match &body { + RequeueRequest::All { .. } => (RequeueScope::All, "crawler", None), + RequeueRequest::Manga { manga_id } => (RequeueScope::Manga(*manga_id), "manga", Some(*manga_id)), + RequeueRequest::Chapter { chapter_id } => { + (RequeueScope::Chapter(*chapter_id), "chapter", Some(*chapter_id)) + } + RequeueRequest::Job { job_id } => (RequeueScope::Job(*job_id), "crawler_job", Some(*job_id)), + }; + let requeued = repo::crawler::requeue_dead_jobs(&state.db, scope).await?; + repo::admin_audit::insert( + &state.db, + admin.0.id, + "crawler_dead_jobs_requeue", + target_kind, + target_id, + json!({ "requeued": requeued, "scope": scope_label(&body) }), + ) + .await?; + Ok(Json(RequeueResponse { requeued })) +} + +fn scope_label(r: &RequeueRequest) -> &'static str { + match r { + RequeueRequest::All { .. } => "all", + RequeueRequest::Manga { .. } => "manga", + RequeueRequest::Chapter { .. } => "chapter", + RequeueRequest::Job { .. } => "job", + } +} diff --git a/backend/src/api/admin/crawler/mod.rs b/backend/src/api/admin/crawler/mod.rs new file mode 100644 index 0000000..47dd037 --- /dev/null +++ b/backend/src/api/admin/crawler/mod.rs @@ -0,0 +1,49 @@ +//! Admin-only crawler observability + control endpoints. +//! +//! Mounted under `/api/v1/admin/crawler*`, cookie-only via `RequireAdmin`. +//! All control endpoints return 503 when the crawler daemon is disabled +//! (`AppState.crawler == None`). Reads compose the live in-process status +//! ([`crate::crawler::status`]) with DB-derived queue counts and the +//! session/browser flags. +//! +//! Split into four siblings for navigability — the surface area grew +//! past the point where keeping it in one file made review harder: +//! - [`status`] — SSE stream + composed status snapshot +//! - [`control`] — run / restart / session endpoints +//! - [`dead_jobs`] — dead-letter list + requeue +//! - [`backlog`] — pending-chapters and missing-covers backlog reads + +mod backlog; +mod control; +mod dead_jobs; +mod status; + +use axum::Router; + +use crate::app::{AppState, CrawlerControl}; +use crate::error::AppError; + +pub fn routes() -> Router { + Router::new() + .merge(status::routes()) + .merge(control::routes()) + .merge(dead_jobs::routes()) + .merge(backlog::routes()) +} + +/// Default page size for the backlog list endpoints. +pub(super) fn default_limit() -> i64 { + 50 +} + +/// Shared 503 helper: the daemon-only control + observability endpoints +/// gate on `AppState.crawler == Some(_)`. Returns the same code and body +/// across every caller so the frontend can rely on `service_unavailable` +/// rather than each handler returning its own variant. +pub(super) fn require_crawler( + state: &AppState, +) -> Result<&std::sync::Arc, AppError> { + state.crawler.as_ref().ok_or_else(|| { + AppError::ServiceUnavailable("crawler daemon is disabled".into()) + }) +} diff --git a/backend/src/api/admin/crawler/status.rs b/backend/src/api/admin/crawler/status.rs new file mode 100644 index 0000000..8fe1255 --- /dev/null +++ b/backend/src/api/admin/crawler/status.rs @@ -0,0 +1,245 @@ +//! GET /admin/crawler — composed status snapshot +//! GET /admin/crawler/stream — Server-Sent Events live status +//! +//! The composed response merges the in-process status surface +//! ([`crate::crawler::status`]) with two DB-derived counts +//! (job-state breakdown and missing-cover backlog) so the dashboard +//! reads the same shape from one one-shot fetch and from each SSE +//! frame. The streaming path debounces a burst of pokes into a single +//! frame and memoizes the DB counts for the [`QUEUE_MEMO_TTL`] window +//! to avoid hammering Postgres. + +use std::convert::Infallible; +use std::time::Duration; + +use axum::extract::State; +use axum::response::sse::{Event, KeepAlive, Sse}; +use axum::routing::get; +use axum::{Json, Router}; +use futures_util::stream::Stream; +use serde::Serialize; + +use crate::app::AppState; +use crate::auth::extractor::RequireAdmin; +use crate::crawler::browser_manager::RestartPhase; +use crate::crawler::status::{ActiveChapter, CoverTarget, LastPass, Phase}; +use crate::error::AppResult; +use crate::repo; + +/// Backstop recompose interval for the SSE stream. Phase/worker/session +/// changes push instantly via the status `watch`; this only bounds the +/// staleness of DB-derived queue counts and the browser phase when those +/// change without an accompanying status poke. +const SSE_BACKSTOP: Duration = Duration::from_secs(5); + +/// Coalesce a burst of status pokes (e.g. one per stored page during a +/// chapter download) into a single SSE frame. After the first change +/// fires we sleep this long and absorb any further changes that arrive +/// in the window before emitting. Tighter than human-perceivable jitter +/// (~16-100 ms is the rule of thumb for "instant"). +const SSE_DEBOUNCE: Duration = Duration::from_millis(250); + +/// Per-connection memo TTL for DB-derived queue counts. With a busy +/// chapter pass bumping the watch up to several times per second, the +/// memo collapses N stream wakeups into one round-trip per second +/// — typically a ~10x reduction in steady-state DB QPS per subscriber. +const QUEUE_MEMO_TTL: Duration = Duration::from_millis(1000); + +pub(super) fn routes() -> Router { + Router::new() + .route("/admin/crawler", get(get_status)) + .route("/admin/crawler/stream", get(stream_status)) +} + +#[derive(Debug, Serialize)] +struct QueueCounts { + pending: i64, + running: i64, + dead: i64, +} + +#[derive(Debug, Serialize)] +struct SessionStatus { + /// Whether the sticky session-expired flag is set (chapter workers idle). + expired: bool, + /// Whether a PHPSESSID is currently configured at all. + configured: bool, +} + +#[derive(Debug, Serialize)] +struct CrawlerStatusResponse { + /// `"running"` | `"disabled"`. + daemon: &'static str, + phase: Option, + /// Configured chapter-worker count (for "N busy / M workers"). + worker_count: usize, + /// Chapters being crawled right now, with live page counts. + active_chapters: Vec, + /// The cover being fetched right now, if any. + current_cover: Option, + /// Mangas still queued for a cover fetch. + covers_queued: i64, + last_pass: LastPass, + session: SessionStatus, + /// `"healthy"` | `"draining"` | `"restarting"` | `"down"`. + browser: &'static str, + queue: QueueCounts, +} + +/// Per-stream memo of the two DB-derived counts shared by every SSE +/// frame: crawler-job state breakdown and missing-cover backlog. Held +/// across iterations of the unfold loop so a burst of status pokes +/// emits one frame from cached counts. A fresh memo (`new`) always +/// misses on first call, so `get_status` (one-shot) sees no stale data. +struct QueueCountsMemo { + cached: Option<(std::time::Instant, (i64, i64, i64), i64)>, +} + +impl QueueCountsMemo { + fn new() -> Self { + Self { cached: None } + } + + async fn get( + &mut self, + db: &sqlx::PgPool, + ) -> sqlx::Result<((i64, i64, i64), i64)> { + if let Some((at, qc, cv)) = self.cached.as_ref() { + if at.elapsed() < QUEUE_MEMO_TTL { + return Ok((*qc, *cv)); + } + } + let qc = repo::crawler::job_state_counts(db).await?; + let cv = repo::crawler::count_missing_covers(db).await?; + self.cached = Some((std::time::Instant::now(), qc, cv)); + Ok((qc, cv)) + } +} + +fn browser_phase_str(p: RestartPhase) -> &'static str { + match p { + RestartPhase::Healthy => "healthy", + RestartPhase::Draining => "draining", + RestartPhase::Restarting => "restarting", + } +} + +/// Compose a full status snapshot from the in-memory status, the +/// browser/session flags, and DB queue-count queries (routed through +/// a memo so a burst of pokes doesn't hammer Postgres). Shared by +/// `get_status` (with a fresh per-call memo) and `stream_status` +/// (with a per-stream memo held across iterations). +async fn compose_status( + state: &AppState, + memo: &mut QueueCountsMemo, +) -> AppResult { + let ((pending, running, dead), covers_queued) = memo.get(&state.db).await?; + let queue = QueueCounts { + pending, + running, + dead, + }; + + Ok(match state.crawler.as_ref() { + None => CrawlerStatusResponse { + daemon: "disabled", + phase: None, + worker_count: 0, + active_chapters: Vec::new(), + current_cover: None, + covers_queued, + last_pass: LastPass::default(), + session: SessionStatus { + expired: false, + configured: false, + }, + browser: "down", + queue, + }, + Some(c) => { + let snap = c.status.snapshot().await; + CrawlerStatusResponse { + daemon: "running", + phase: Some(snap.phase), + worker_count: snap.worker_count, + active_chapters: snap.active_chapters, + current_cover: snap.current_cover, + covers_queued, + last_pass: snap.last_pass, + session: SessionStatus { + expired: c.session.is_expired(), + configured: c.session.current().await.is_some(), + }, + browser: browser_phase_str(c.browser_manager.phase()), + queue, + } + } + }) +} + +async fn get_status( + State(state): State, + _admin: RequireAdmin, +) -> AppResult> { + // Fresh memo — one-shot calls always query the DB. The wrapper exists + // only so compose_status can share its signature with the streaming + // path; there is no caching across one-shot calls. + let mut memo = QueueCountsMemo::new(); + Ok(Json(compose_status(&state, &mut memo).await?)) +} + +/// Push live status to the dashboard instead of polling. Emits a snapshot +/// immediately on connect, then on every status change (instant, via the +/// `watch` notifier) and on a [`SSE_BACKSTOP`] tick (to refresh DB queue +/// counts / browser phase that change without a status poke). The browser +/// opens this only while the crawler page is mounted and closes it on +/// navigate-away, so the subscription is scoped to the active page. +async fn stream_status( + State(state): State, + _admin: RequireAdmin, +) -> Sse>> { + // Subscribe before the first emit so no change between the initial + // snapshot and the first await is lost. + let rx = state.crawler.as_ref().map(|c| c.status.subscribe()); + let memo = QueueCountsMemo::new(); + + let stream = futures_util::stream::unfold( + (state, rx, memo, true), + |(state, mut rx, mut memo, first)| async move { + // After the first immediate emit, wait for a change or the + // backstop tick before recomposing. On a change, debounce a + // short window so a burst of pokes (one per stored page + // during a chapter download, etc.) coalesces into a single + // frame instead of hammering subscribers. + if !first { + match rx.as_mut() { + Some(rx) => { + tokio::select! { + _ = rx.changed() => { + tokio::time::sleep(SSE_DEBOUNCE).await; + // Mark any pokes that arrived during the + // debounce window as observed so the next + // iteration only fires on NEW changes. + rx.borrow_and_update(); + } + _ = tokio::time::sleep(SSE_BACKSTOP) => {} + } + } + None => tokio::time::sleep(SSE_BACKSTOP).await, + } + } + // Compose; on a transient DB error, emit a keep-alive comment + // rather than tearing down the stream. + let event = match compose_status(&state, &mut memo).await { + Ok(resp) => Event::default() + .event("status") + .json_data(&resp) + .unwrap_or_else(|_| Event::default().comment("serialize error")), + Err(_) => Event::default().comment("status unavailable"), + }; + Some((Ok(event), (state, rx, memo, false))) + }, + ); + + Sse::new(stream).keep_alive(KeepAlive::default()) +} diff --git a/backend/src/api/admin/mod.rs b/backend/src/api/admin/mod.rs index 0154665..2048cf1 100644 --- a/backend/src/api/admin/mod.rs +++ b/backend/src/api/admin/mod.rs @@ -4,6 +4,7 @@ //! bot/API tokens cannot reach admin routes (see //! `crate::auth::extractor::RequireAdmin`). +pub mod crawler; pub mod mangas; pub mod resync; pub mod system; @@ -19,4 +20,5 @@ pub fn routes() -> Router { .merge(mangas::routes()) .merge(resync::routes()) .merge(system::routes()) + .merge(crawler::routes()) } diff --git a/backend/src/app.rs b/backend/src/app.rs index 18d1192..b5250e6 100644 --- a/backend/src/app.rs +++ b/backend/src/app.rs @@ -1,5 +1,5 @@ use std::sync::Arc; -use std::sync::atomic::AtomicBool; +use std::sync::atomic::{AtomicBool, AtomicU32, Ordering}; use anyhow::Context; use async_trait::async_trait; @@ -46,6 +46,38 @@ pub struct AppState { /// same wiring that builds the daemon's chapter dispatcher, so a /// force resync uses the daemon's BrowserManager + rate limiters. pub resync: Option>, + /// Crawler observability + control handle (live status, coordinated + /// browser restart, runtime session, manual run). `None` when the + /// daemon is disabled; admin handlers gate on `.is_some()` → 503. + pub crawler: Option>, + /// Browser origins permitted to issue mutating requests to + /// `/api/v1/admin/*`. See [`crate::config::Config::admin_allowed_origins`] + /// for the policy. Cloned per-request into the CSRF middleware; the + /// `Arc` keeps the clone cheap. Empty list → check is skipped + /// (operator opt-out documented in `.env.example`). + pub admin_allowed_origins: Arc>, +} + +/// Shared handle the admin crawler endpoints use to observe and control +/// the running daemon. Bundled so the handlers take one optional field on +/// `AppState` rather than many. +pub struct CrawlerControl { + pub browser_manager: Arc, + pub session: Arc, + pub status: crate::crawler::status::StatusHandle, + /// Used by the "run metadata pass now" endpoint; `None` when no + /// `CRAWLER_START_URL` is configured (cron disabled). + pub metadata_pass: Option>, + /// Drain budget for a manually-triggered coordinated browser restart. + pub drain_deadline: std::time::Duration, + /// Held for the duration of a `/admin/crawler/run` pass so a second + /// click returns 409 instead of fanning N overlapping metadata passes + /// onto the single browser lease. The daemon's daily cron does NOT + /// take this lock — cron and operator-triggered are different + /// trust modes (cron is single-fire by definition; the lock only + /// dedups operator clicks). `Arc` so the spawned task can hold an + /// owned guard past the request boundary. + pub manual_pass_lock: Arc>, } /// Bundle returned by [`build`]. The router is what `axum::serve` consumes; @@ -80,12 +112,12 @@ pub async fn build(config: Config) -> anyhow::Result { let storage: Arc = Arc::new(LocalStorage::new(config.storage_dir.clone())); - let (daemon, resync) = if config.crawler.daemon_enabled { + let (daemon, resync, crawler) = if config.crawler.daemon_enabled { let spawned = spawn_crawler_daemon(db.clone(), Arc::clone(&storage), &config.crawler).await?; - (Some(spawned.handle), Some(spawned.resync)) + (Some(spawned.handle), Some(spawned.resync), Some(spawned.crawler)) } else { tracing::info!("crawler daemon disabled (CRAWLER_DAEMON=false)"); - (None, None) + (None, None, None) }; let auth_limiter = Arc::new(AuthRateLimiter::new(config.auth.rate_limit)); @@ -96,6 +128,8 @@ pub async fn build(config: Config) -> anyhow::Result { upload: config.upload.clone(), auth_limiter, resync, + crawler, + admin_allowed_origins: Arc::new(config.admin_allowed_origins.clone()), }; let router = router(state).layer(cors_layer(&config.cors_allowed_origins)); Ok(AppHandle { router, daemon }) @@ -108,6 +142,7 @@ pub async fn build(config: Config) -> anyhow::Result { struct SpawnedDaemon { handle: daemon::DaemonHandle, resync: Arc, + crawler: Arc, } async fn spawn_crawler_daemon( @@ -115,11 +150,17 @@ async fn spawn_crawler_daemon( storage: Arc, cfg: &CrawlerConfig, ) -> anyhow::Result { - // Reqwest client with cookie jar pre-seeded so CDN image fetches - // include PHPSESSID. Same shape as bin/crawler.rs main(). + // Reqwest client with a shared cookie jar so CDN image fetches include + // PHPSESSID. The same `Arc` is held by the SessionController, so a + // runtime session refresh rewrites it in place. Initial value: a + // persisted runtime session (survives restart) takes precedence over + // CRAWLER_PHPSESSID env. let cookie_jar = Arc::new(reqwest::cookie::Jar::default()); + let initial_sid = crate::crawler::session_control::SessionController::load_persisted(&db) + .await + .or_else(|| cfg.phpsessid.clone()); if let (Some(sid), Some(domain), Some(start_url)) = - (&cfg.phpsessid, &cfg.cookie_domain, &cfg.start_url) + (&initial_sid, &cfg.cookie_domain, &cfg.start_url) { let cookie_str = format!("PHPSESSID={sid}; Domain={domain}; Path=/"); let seed_url = reqwest::Url::parse(start_url) @@ -129,7 +170,7 @@ async fn spawn_crawler_daemon( let mut http_builder = reqwest::Client::builder() .timeout(std::time::Duration::from_secs(30)) .no_proxy() - .cookie_provider(cookie_jar); + .cookie_provider(Arc::clone(&cookie_jar)); if let Some(ua) = &cfg.user_agent { http_builder = http_builder.user_agent(ua); } @@ -157,6 +198,23 @@ async fn spawn_crawler_daemon( } let tor_recircuit_max = cfg.tor_recircuit_max_attempts; + // Session controller + sticky session-expired flag. Created before the + // browser so the on_launch hook can read the *current* session value + // (rather than a value captured at startup), and so a runtime refresh + // updates the cookie everywhere. + let session_expired = Arc::new(AtomicBool::new(false)); + let session_controller = crate::crawler::session_control::SessionController::new( + initial_sid, + Arc::clone(&cookie_jar), + cfg.cookie_domain.clone(), + cfg.start_url.clone(), + db.clone(), + Arc::clone(&session_expired), + ); + + // Live status surface, sized to the worker count. + let status = crate::crawler::status::StatusHandle::new(cfg.chapter_workers); + // Browser manager. on_launch re-injects PHPSESSID on every fresh // chromium spawn so an idle teardown followed by re-launch stays // authenticated without operator action. @@ -165,18 +223,25 @@ async fn spawn_crawler_daemon( let chromium_proxy = crate::crawler::url_utils::chromium_proxy_arg(proxy); launch_opts.extra_args.push(format!("--proxy-server={chromium_proxy}")); } - let on_launch = match (&cfg.phpsessid, &cfg.cookie_domain, &cfg.start_url) { - (Some(sid), Some(domain), Some(start_url)) => { - let sid = sid.clone(); + let on_launch = match (&cfg.cookie_domain, &cfg.start_url) { + (Some(domain), Some(start_url)) => { let domain = domain.clone(); let start_url = start_url.clone(); let tor_for_launch = tor.as_ref().map(Arc::clone); + let sc = Arc::clone(&session_controller); let on_launch: browser_manager::OnLaunch = Arc::new(move |browser| { - let sid = sid.clone(); let domain = domain.clone(); let start_url = start_url.clone(); let tor_for_launch = tor_for_launch.as_ref().map(Arc::clone); + let sc = Arc::clone(&sc); Box::pin(async move { + // Read the *current* session each launch so a runtime + // refresh is picked up on the next (re)launch. No session + // configured → run unauthenticated (metadata needs no auth). + let Some(sid) = sc.current().await else { + tracing::info!("on_launch: no session set — skipping inject + probe"); + return Ok(()); + }; session::inject_phpsessid(&browser, &sid, &domain) .await .context("on_launch: inject_phpsessid")?; @@ -197,8 +262,6 @@ async fn spawn_crawler_daemon( }; let browser_manager = BrowserManager::new(launch_opts, cfg.idle_timeout, on_launch); - let session_expired = Arc::new(AtomicBool::new(false)); - let metadata_pass: Option> = cfg.start_url.as_ref().map(|url| { let m: Arc = Arc::new(RealMetadataPass { browser_manager: Arc::clone(&browser_manager), @@ -210,6 +273,8 @@ async fn spawn_crawler_daemon( manga_limit: cfg.manga_limit, download_allowlist: cfg.download_allowlist.clone(), max_image_bytes: cfg.max_image_bytes, + metadata_max_consecutive_failures: cfg.metadata_max_consecutive_failures, + status: status.clone(), tor: tor.as_ref().map(Arc::clone), }); m @@ -223,6 +288,10 @@ async fn spawn_crawler_daemon( rate: Arc::clone(&rate), download_allowlist: cfg.download_allowlist.clone(), max_image_bytes: cfg.max_image_bytes, + transient_failures: Arc::new(AtomicU32::new(0)), + restart_threshold: cfg.browser_restart_threshold, + drain_deadline: cfg.job_timeout, + status: status.clone(), tor: tor.as_ref().map(Arc::clone), }); @@ -260,20 +329,32 @@ async fn spawn_crawler_daemon( db, cancel, DaemonConfig { - metadata_pass, + metadata_pass: metadata_pass.clone(), dispatcher, chapter_workers: cfg.chapter_workers, daily_at: cfg.daily_at, tz: cfg.tz, retention_days: cfg.retention_days, session_expired, + status: status.clone(), + job_timeout: cfg.job_timeout, extra_tasks: vec![reaper_task, shutdown_task], }, ); + let crawler = Arc::new(CrawlerControl { + browser_manager: Arc::clone(&browser_manager), + session: session_controller, + status, + metadata_pass, + drain_deadline: cfg.job_timeout, + manual_pass_lock: Arc::new(tokio::sync::Mutex::new(())), + }); + Ok(SpawnedDaemon { handle: daemon_handle, resync, + crawler, }) } @@ -292,6 +373,8 @@ struct RealMetadataPass { manga_limit: usize, download_allowlist: DownloadAllowlist, max_image_bytes: usize, + metadata_max_consecutive_failures: u32, + status: crate::crawler::status::StatusHandle, tor: Option>, } @@ -309,6 +392,8 @@ impl MetadataPass for RealMetadataPass { false, &self.download_allowlist, self.max_image_bytes, + self.metadata_max_consecutive_failures, + Some(&self.status), self.tor.as_deref(), ) .await; @@ -321,7 +406,8 @@ impl MetadataPass for RealMetadataPass { // errored — the early-stop walk can complete its work and bail // late, and a transient browser failure shouldn't cancel the // residual cover backlog. The backfill has its own per-call cap - // so a runaway error stream can't monopolise the tick. + // so a runaway error stream can't monopolise the tick. It sets the + // CoverBackfill{index,total} phase + current_cover per entry. match pipeline::backfill_missing_covers( &self.browser_manager, &self.db, @@ -331,6 +417,7 @@ impl MetadataPass for RealMetadataPass { pipeline::COVER_BACKFILL_DEFAULT_MAX, &self.download_allowlist, self.max_image_bytes, + Some(&self.status), self.tor.as_deref(), ) .await @@ -359,6 +446,16 @@ struct RealChapterDispatcher { rate: Arc, download_allowlist: DownloadAllowlist, max_image_bytes: usize, + /// Consecutive transient chapter failures; resets on any success. + /// Drives the automatic coordinated browser restart. + transient_failures: Arc, + /// Consecutive-failure count that triggers an auto restart. + restart_threshold: u32, + /// How long a coordinated restart waits for in-flight leases to drain. + drain_deadline: std::time::Duration, + /// Live status surface — the dispatcher registers each chapter it + /// crawls (with a realtime page count) here. + status: crate::crawler::status::StatusHandle, tor: Option>, } @@ -374,10 +471,21 @@ impl ChapterDispatcher for RealChapterDispatcher { let row = repo::chapter::dispatch_target(&self.db, chapter_id) .await .context("look up chapter for dispatch")?; - let Some((manga_id, source_url)) = row else { + let Some((manga_id, source_url, manga_title, chapter_number)) = row else { // Chapter (or its source row) is gone — ack done. return Ok(SyncOutcome::Skipped); }; + // Register the chapter as crawling now (live status). The + // guard removes it on every exit path — success, panic, or + // the worker's outer-timeout drop. + let _active = self.status.begin_chapter(crate::crawler::status::ActiveChapter { + manga_id, + manga_title, + chapter_id, + chapter_number, + pages_done: 0, + pages_total: None, + }); let lease = self.browser_manager.acquire().await?; let result = content::sync_chapter_content( &lease, @@ -392,14 +500,37 @@ impl ChapterDispatcher for RealChapterDispatcher { &self.download_allowlist, self.max_image_bytes, self.tor.as_deref(), + Some(&self.status), ) .await; drop(lease); match result { - Ok(outcome) => Ok(outcome), + Ok(outcome) => { + // Any successful dispatch (including a clean Skipped) + // means the browser is healthy — reset the streak. + self.transient_failures.store(0, Ordering::Release); + Ok(outcome) + } Err(e) => { + let streak = self.transient_failures.fetch_add(1, Ordering::AcqRel) + 1; if crate::crawler::nav::anyhow_looks_browser_dead(&e) { + // Hard browser-dead: lazy invalidate (next acquire + // relaunches). Reset the streak — we're recovering. self.browser_manager.invalidate().await; + self.transient_failures.store(0, Ordering::Release); + } else if self.restart_threshold > 0 && streak >= self.restart_threshold { + // Persistent transients that TOR recircuit couldn't + // fix — proactively restart Chromium. + tracing::warn!( + streak, + threshold = self.restart_threshold, + "auto browser restart: consecutive transient chapter failures" + ); + let _ = self + .browser_manager + .coordinated_restart(self.drain_deadline) + .await; + self.transient_failures.store(0, Ordering::Release); } Err(e) } @@ -419,6 +550,11 @@ pub fn router(state: AppState) -> Router { let max_request_bytes = state.upload.max_request_bytes; Router::new() .nest("/api/v1", crate::api::routes()) + .layer(middleware::from_fn(admin_no_store_guard)) + .layer(middleware::from_fn_with_state( + state.clone(), + admin_csrf_guard, + )) .layer(middleware::from_fn_with_state( state.clone(), private_mode_guard, @@ -428,6 +564,113 @@ pub fn router(state: AppState) -> Router { .layer(TraceLayer::new_for_http()) } +/// Path prefix the admin-only middlewares scope themselves to. The router +/// already nests `/api/v1`, so callers see `/api/v1/admin/...`. +const ADMIN_PATH_PREFIX: &str = "/api/v1/admin/"; + +/// CSRF defence for cookie-authenticated admin mutations. The session +/// cookie is `SameSite=Lax`, which still permits top-level form-POSTs +/// from a malicious page — this middleware rejects such requests by +/// comparing the request's `Origin` (with `Referer` as fallback) against +/// the configured allowlist. Safe methods (`GET`/`HEAD`/`OPTIONS`) are +/// always allowed. Requests with neither `Origin` nor `Referer` are +/// allowed (non-browser callers like curl can't be a CSRF vector). When +/// the allowlist is empty the check is skipped entirely (operator +/// opt-out — documented in `.env.example`). +async fn admin_csrf_guard( + State(state): State, + req: Request, + next: Next, +) -> Result { + if !req.uri().path().starts_with(ADMIN_PATH_PREFIX) { + return Ok(next.run(req).await); + } + if matches!( + *req.method(), + Method::GET | Method::HEAD | Method::OPTIONS + ) { + return Ok(next.run(req).await); + } + if state.admin_allowed_origins.is_empty() { + return Ok(next.run(req).await); + } + let headers = req.headers(); + let origin = headers.get("origin").and_then(|v| v.to_str().ok()); + let referer = headers.get("referer").and_then(|v| v.to_str().ok()); + // No Origin AND no Referer → server-to-server / curl / extension. + // Browsers always send one or the other on a cross-site POST. + let Some(candidate) = origin.or(referer) else { + return Ok(next.run(req).await); + }; + if origin_in_allowlist(candidate, &state.admin_allowed_origins) { + return Ok(next.run(req).await); + } + tracing::warn!( + candidate = %truncate_for_log(candidate, 64), + path = %req.uri().path(), + "admin CSRF: rejecting mutation with disallowed origin" + ); + Err(AppError::Forbidden) +} + +/// Match `candidate` (an `Origin` value, or a `Referer` URL whose +/// origin we'll extract) against `allowed`. `Origin` is `scheme://host[:port]` +/// with no path; `Referer` is a full URL — compare by parsing both and +/// matching scheme + host + port. +fn origin_in_allowlist(candidate: &str, allowed: &[String]) -> bool { + let cand_origin = parse_origin(candidate); + let Some(cand) = cand_origin else { return false }; + allowed + .iter() + .filter_map(|a| parse_origin(a)) + .any(|a| a == cand) +} + +/// Extract the origin (`scheme://host[:port]`) from an `Origin` header +/// value or a `Referer` URL. Returns `None` when the input doesn't parse +/// as a URL with a host. +fn parse_origin(raw: &str) -> Option { + let url = reqwest::Url::parse(raw.trim()).ok()?; + let host = url.host_str()?; + let scheme = url.scheme(); + let port_str = match (url.port(), scheme) { + (Some(p), "http") if p == 80 => String::new(), + (Some(p), "https") if p == 443 => String::new(), + (Some(p), _) => format!(":{p}"), + (None, _) => String::new(), + }; + Some(format!("{scheme}://{host}{port_str}")) +} + +fn truncate_for_log(s: &str, max: usize) -> &str { + let end = s + .char_indices() + .take(max) + .last() + .map(|(i, c)| i + c.len_utf8()) + .unwrap_or(0); + &s[..end.min(s.len())] +} + +/// Forbids intermediaries (CDN, browser bfcache, reverse proxy with a +/// permissive default) from caching admin responses. Defence-in-depth +/// for cookie-authenticated reads — even though the responses already +/// vary on cookie, a misconfigured cache layer in front of the +/// SvelteKit container could leak a logged-in admin's view to another +/// session. Headers added on response so the rest of the API is +/// unaffected. +async fn admin_no_store_guard(req: Request, next: Next) -> Response { + let is_admin_path = req.uri().path().starts_with(ADMIN_PATH_PREFIX); + let mut resp = next.run(req).await; + if is_admin_path { + resp.headers_mut().insert( + axum::http::header::CACHE_CONTROL, + HeaderValue::from_static("no-store"), + ); + } + resp +} + /// Paths reachable anonymously even when `PRIVATE_MODE=true`. Login and /// logout are needed for the auth flow itself; `/health` is reserved /// for load-balancer probes; `/auth/config` lets the frontend decide diff --git a/backend/src/bin/crawler.rs b/backend/src/bin/crawler.rs index 15ca8d4..f36b3b8 100644 --- a/backend/src/bin/crawler.rs +++ b/backend/src/bin/crawler.rs @@ -303,6 +303,11 @@ async fn run( skip_chapters, allowlist.as_ref(), max_image_bytes, + // Circuit-breaker disabled for the operator-driven CLI: a manual + // sweep should push through transient failures, not self-abort. + 0, + // No live status surface for the one-shot CLI. + None, tor.as_deref(), ) .await?; @@ -412,6 +417,8 @@ async fn sync_bookmarked_chapter_content( allowlist.as_ref(), max_image_bytes, tor.as_deref(), + // CLI one-shot — no live status surface. + None, ) .await; drop(lease); diff --git a/backend/src/config.rs b/backend/src/config.rs index b54aadc..37effaf 100644 --- a/backend/src/config.rs +++ b/backend/src/config.rs @@ -74,6 +74,20 @@ pub struct Config { pub auth: AuthConfig, pub upload: UploadConfig, pub cors_allowed_origins: Vec, + /// Origins (scheme + host[:port]) that may issue browser-driven + /// mutating requests to `/api/v1/admin/*`. Defends against + /// SameSite=Lax CSRF on the admin cookie: a top-level form POST from + /// a malicious page still carries the cookie, but the middleware + /// rejects it when the `Origin` (or `Referer` fallback) is absent + /// from this list. Sourced from `ADMIN_ALLOWED_ORIGINS` + /// (comma-separated). Leave empty to skip the check entirely + /// (curl / server-to-server callers send neither header, so they + /// pass; same-origin browser requests don't have Origin set on + /// same-origin POSTs in some browsers either — operators on a + /// same-origin deploy can leave this empty, but doing so removes + /// the CSRF defence). Safe methods (GET/HEAD/OPTIONS) never trigger + /// the check. + pub admin_allowed_origins: Vec, pub crawler: CrawlerConfig, /// `(username, password)` for the admin user provisioned at startup /// when both `ADMIN_USERNAME` and `ADMIN_PASSWORD` are set. `None` @@ -132,6 +146,19 @@ pub struct CrawlerConfig { /// (full sweep up to the source's own bound). Sourced from /// `CRAWLER_LIMIT`, mirroring the CLI binary. pub manga_limit: usize, + /// Hard upper bound on a single chapter-content job dispatch. A job + /// exceeding this is acked failed (exponential backoff) instead of + /// wedging a worker. Defaults to 600s. `CRAWLER_JOB_TIMEOUT_SECS`. + pub job_timeout: Duration, + /// Consecutive `fetch_manga` failures that abort a metadata pass + /// (circuit-breaker for a source outage). The pass does NOT mark a + /// clean exit, so the next tick does a recovery sweep. Defaults to + /// 10. `CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES`. + pub metadata_max_consecutive_failures: u32, + /// Consecutive transient chapter failures (after TOR recircuit is + /// exhausted) that trigger an automatic coordinated browser restart. + /// Defaults to 3. `CRAWLER_BROWSER_RESTART_THRESHOLD`. + pub browser_restart_threshold: u32, } impl Default for CrawlerConfig { @@ -159,6 +186,9 @@ impl Default for CrawlerConfig { download_allowlist: DownloadAllowlist::new(), max_image_bytes: DEFAULT_MAX_IMAGE_BYTES, manga_limit: 0, + job_timeout: Duration::from_secs(600), + metadata_max_consecutive_failures: 10, + browser_restart_threshold: 3, } } } @@ -205,6 +235,15 @@ impl Config { .collect() }) .unwrap_or_default(), + admin_allowed_origins: std::env::var("ADMIN_ALLOWED_ORIGINS") + .ok() + .map(|s| { + s.split(',') + .map(|o| o.trim().to_string()) + .filter(|o| !o.is_empty()) + .collect() + }) + .unwrap_or_default(), crawler: CrawlerConfig::from_env()?, admin_bootstrap: admin_bootstrap_from_env(), }) @@ -283,6 +322,13 @@ impl CrawlerConfig { download_allowlist, max_image_bytes: env_usize("CRAWLER_MAX_IMAGE_BYTES", DEFAULT_MAX_IMAGE_BYTES), manga_limit: env_usize("CRAWLER_LIMIT", 0), + job_timeout: Duration::from_secs(env_u64("CRAWLER_JOB_TIMEOUT_SECS", 600).max(1)), + metadata_max_consecutive_failures: env_u64( + "CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES", + 10, + ) as u32, + browser_restart_threshold: env_u64("CRAWLER_BROWSER_RESTART_THRESHOLD", 3).max(1) + as u32, }) } } @@ -384,6 +430,33 @@ mod tests { assert_eq!(cfg.manga_limit, 0); } + #[test] + fn reliability_knobs_default_when_unset() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + std::env::remove_var("CRAWLER_JOB_TIMEOUT_SECS"); + std::env::remove_var("CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES"); + std::env::remove_var("CRAWLER_BROWSER_RESTART_THRESHOLD"); + let cfg = CrawlerConfig::from_env().expect("from_env"); + assert_eq!(cfg.job_timeout, Duration::from_secs(600)); + assert_eq!(cfg.metadata_max_consecutive_failures, 10); + assert_eq!(cfg.browser_restart_threshold, 3); + } + + #[test] + fn reliability_knobs_parse_from_env() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + std::env::set_var("CRAWLER_JOB_TIMEOUT_SECS", "120"); + std::env::set_var("CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES", "5"); + std::env::set_var("CRAWLER_BROWSER_RESTART_THRESHOLD", "7"); + let cfg = CrawlerConfig::from_env().expect("from_env"); + std::env::remove_var("CRAWLER_JOB_TIMEOUT_SECS"); + std::env::remove_var("CRAWLER_METADATA_MAX_CONSECUTIVE_FAILURES"); + std::env::remove_var("CRAWLER_BROWSER_RESTART_THRESHOLD"); + assert_eq!(cfg.job_timeout, Duration::from_secs(120)); + assert_eq!(cfg.metadata_max_consecutive_failures, 5); + assert_eq!(cfg.browser_restart_threshold, 7); + } + #[test] fn private_mode_env_parses_true() { let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); diff --git a/backend/src/crawler/browser_manager.rs b/backend/src/crawler/browser_manager.rs index c1725e4..47c5ac0 100644 --- a/backend/src/crawler/browser_manager.rs +++ b/backend/src/crawler/browser_manager.rs @@ -13,7 +13,7 @@ //! until [`BrowserManager::shutdown`]. use std::ops::Deref; -use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU8, AtomicUsize, Ordering}; use std::sync::Arc; use std::time::Duration; @@ -71,12 +71,42 @@ impl ActiveTracker { } } +/// Lifecycle gate for a coordinated browser restart. `acquire()` parks +/// while not [`RestartPhase::Healthy`] so no new navigation starts mid- +/// restart; long-lived lease holders (the metadata pass) cooperate by +/// checking [`BrowserManager::is_restart_pending`] at safe boundaries. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum RestartPhase { + /// Normal operation — acquires proceed. + Healthy, + /// Restart requested; new acquires park, waiting for in-flight leases + /// to drain. + Draining, + /// Chromium is being closed + relaunched. + Restarting, +} + +const PHASE_HEALTHY: u8 = 0; +const PHASE_DRAINING: u8 = 1; +const PHASE_RESTARTING: u8 = 2; + pub struct BrowserManager { inner: Mutex, active: Arc, launch_opts: LaunchOptions, idle_timeout: Duration, on_launch: OnLaunch, + /// Coarse lifecycle phase (one of the `PHASE_*` constants). + phase: AtomicU8, + /// Woken when the phase returns to `Healthy` so parked acquires resume. + resume: Notify, + /// Serialises coordinated restarts so concurrent requests collapse into + /// a single relaunch. + restart_lock: Mutex<()>, + /// Result of the most recent relaunch, so a caller that coalesced into + /// an in-progress restart reports that restart's real outcome instead + /// of a blind success. + last_restart_ok: AtomicBool, } struct Inner { @@ -99,28 +129,72 @@ impl BrowserManager { launch_opts, idle_timeout, on_launch, + phase: AtomicU8::new(PHASE_HEALTHY), + resume: Notify::new(), + restart_lock: Mutex::new(()), + last_restart_ok: AtomicBool::new(true), }) } + /// Current restart phase. + pub fn phase(&self) -> RestartPhase { + match self.phase.load(Ordering::Acquire) { + PHASE_DRAINING => RestartPhase::Draining, + PHASE_RESTARTING => RestartPhase::Restarting, + _ => RestartPhase::Healthy, + } + } + + fn set_phase(&self, phase: RestartPhase) { + let v = match phase { + RestartPhase::Healthy => PHASE_HEALTHY, + RestartPhase::Draining => PHASE_DRAINING, + RestartPhase::Restarting => PHASE_RESTARTING, + }; + self.phase.store(v, Ordering::Release); + } + + /// Whether a coordinated restart is in progress. Long-lived lease + /// holders poll this at safe boundaries and yield their lease so the + /// drain can complete promptly. + pub fn is_restart_pending(&self) -> bool { + self.phase() != RestartPhase::Healthy + } + + /// Launch Chromium into `guard`, running the `on_launch` hook before + /// publishing the handle so a probe failure doesn't leave a half- + /// initialised browser behind. + async fn launch_into(&self, guard: &mut Inner) -> anyhow::Result<()> { + let handle = browser::launch(self.launch_opts.clone()) + .await + .context("BrowserManager: launch chromium")?; + let shared = handle.shared(); + if let Err(e) = (self.on_launch)(Arc::clone(&shared)).await { + let _ = handle.close().await; + return Err(e.context("BrowserManager: on_launch hook failed")); + } + guard.handle = Some(handle); + guard.shared = Some(shared); + Ok(()) + } + /// Acquire a shared browser lease. The first acquire after a teardown /// launches a fresh Chromium (and runs `on_launch`); subsequent acquires /// while a process is alive just bump the counter and clone the `Arc`. pub async fn acquire(&self) -> anyhow::Result { + // Park while a coordinated restart is draining/relaunching so no new + // navigation starts against a browser that's about to be torn down. + // The short sleep fallback guarantees liveness even if a `resume` + // notification is missed (classic Notify lost-wakeup). + while self.phase() != RestartPhase::Healthy { + tokio::select! { + _ = self.resume.notified() => {} + _ = tokio::time::sleep(Duration::from_millis(100)) => {} + } + } let mut guard = self.inner.lock().await; if guard.handle.is_none() { - let handle = browser::launch(self.launch_opts.clone()) - .await - .context("BrowserManager: launch chromium")?; - let shared = handle.shared(); - // Run the on-launch hook before publishing the handle so a session - // probe failure doesn't leave a half-initialized browser behind. - if let Err(e) = (self.on_launch)(Arc::clone(&shared)).await { - // Close the just-launched browser since we won't be using it. - let _ = handle.close().await; - return Err(e.context("BrowserManager: on_launch hook failed")); - } - guard.handle = Some(handle); - guard.shared = Some(shared); + self.launch_into(&mut guard).await?; } let browser = guard .shared @@ -134,6 +208,51 @@ impl BrowserManager { }) } + /// Coordinated restart: block new acquires, wait for in-flight leases + /// to drain (up to `drain_deadline`, then force), close + relaunch + /// Chromium (re-running `on_launch` → re-inject session + probe), then + /// resume parked acquirers. Concurrent calls collapse into one + /// relaunch. The phase is always returned to `Healthy` — even if the + /// relaunch errors — so a failed restart never permanently wedges + /// acquisition (the next acquire retries the launch lazily). + pub async fn coordinated_restart(&self, drain_deadline: Duration) -> anyhow::Result<()> { + // Dedup: if a restart is already running, wait for it and report + // that restart's real outcome (not a blind success). + let _restart_guard = match self.restart_lock.try_lock() { + Ok(g) => g, + Err(_) => { + let _ = self.restart_lock.lock().await; + return if self.last_restart_ok.load(Ordering::Acquire) { + Ok(()) + } else { + Err(anyhow::anyhow!("a concurrent coordinated browser restart failed")) + }; + } + }; + + self.set_phase(RestartPhase::Draining); + await_drain(&self.active, drain_deadline).await; + + self.set_phase(RestartPhase::Restarting); + let relaunch = { + let mut guard = self.inner.lock().await; + guard.shared = None; + if let Some(handle) = guard.handle.take() { + let _ = handle.close().await; + } + self.launch_into(&mut guard).await + }; + + self.last_restart_ok.store(relaunch.is_ok(), Ordering::Release); + self.set_phase(RestartPhase::Healthy); + self.resume.notify_waiters(); + match &relaunch { + Ok(()) => tracing::info!("BrowserManager: coordinated restart complete"), + Err(e) => tracing::error!(error = ?e, "BrowserManager: coordinated restart relaunch failed"), + } + relaunch.context("coordinated_restart: relaunch") + } + /// Forcefully close the cached browser regardless of active count. /// Used on daemon shutdown. After this returns the next acquire will /// re-launch from scratch. @@ -176,6 +295,29 @@ impl BrowserManager { } } +/// Wait for the active-lease count to reach zero, up to `deadline`. Wakes +/// on the tracker's idle signal and re-checks on a short poll so a missed +/// signal can't strand the drain. Returns when drained or when the +/// deadline elapses (the caller then force-restarts). Extracted as a free +/// fn so the timing logic is unit-testable without launching Chromium. +async fn await_drain(active: &Arc, deadline: Duration) { + let start = tokio::time::Instant::now(); + while active.current() > 0 { + let Some(remaining) = deadline.checked_sub(start.elapsed()) else { + tracing::warn!( + active = active.current(), + "coordinated_restart: drain deadline exceeded — forcing relaunch" + ); + return; + }; + let nap = remaining.min(Duration::from_millis(250)); + tokio::select! { + _ = active.idle_signal().notified() => {} + _ = tokio::time::sleep(nap) => {} + } + } +} + /// Background reaper. Returns immediately when `idle_timeout == 0`. /// Otherwise spawns a task that: /// 1. Waits on `idle_signal` (woken when active hits zero). @@ -270,6 +412,63 @@ mod tests { mgr.invalidate().await; } + #[tokio::test] + async fn await_drain_returns_immediately_when_already_idle() { + let active = ActiveTracker::new(); + let start = tokio::time::Instant::now(); + await_drain(&active, Duration::from_secs(5)).await; + assert!(start.elapsed() < Duration::from_millis(200), "no wait when idle"); + } + + #[tokio::test] + async fn await_drain_completes_when_lease_released() { + let active = ActiveTracker::new(); + active.acquire(); + let bg = { + let a = Arc::clone(&active); + tokio::spawn(async move { + tokio::time::sleep(Duration::from_millis(100)).await; + a.release(); + }) + }; + // Generous deadline; should return shortly after the release, not + // at the deadline. + let start = tokio::time::Instant::now(); + await_drain(&active, Duration::from_secs(5)).await; + assert!(start.elapsed() < Duration::from_secs(2), "drained on release"); + assert_eq!(active.current(), 0); + bg.await.unwrap(); + } + + #[tokio::test] + async fn await_drain_force_returns_after_deadline_when_stuck() { + let active = ActiveTracker::new(); + active.acquire(); // never released + let start = tokio::time::Instant::now(); + await_drain(&active, Duration::from_millis(300)).await; + let elapsed = start.elapsed(); + assert!(elapsed >= Duration::from_millis(250), "waited ~deadline: {elapsed:?}"); + assert!(elapsed < Duration::from_secs(2), "but not forever: {elapsed:?}"); + assert_eq!(active.current(), 1, "still held — caller force-restarts"); + } + + #[test] + fn phase_transitions_reflect_is_restart_pending() { + let mgr = BrowserManager::new( + crate::crawler::browser::LaunchOptions::default(), + Duration::ZERO, + noop_on_launch(), + ); + assert_eq!(mgr.phase(), RestartPhase::Healthy); + assert!(!mgr.is_restart_pending()); + mgr.set_phase(RestartPhase::Draining); + assert!(mgr.is_restart_pending()); + mgr.set_phase(RestartPhase::Restarting); + assert!(mgr.is_restart_pending()); + mgr.set_phase(RestartPhase::Healthy); + assert!(!mgr.is_restart_pending()); + } + #[tokio::test] async fn active_tracker_signals_idle_only_on_zero_transition() { let tracker = ActiveTracker::new(); diff --git a/backend/src/crawler/content.rs b/backend/src/crawler/content.rs index 1549786..13ffa88 100644 --- a/backend/src/crawler/content.rs +++ b/backend/src/crawler/content.rs @@ -18,9 +18,9 @@ use uuid::Uuid; use crate::crawler::detect::PageError; use crate::crawler::rate_limit::HostRateLimiters; -use crate::crawler::safety::{fetch_bytes_capped, looks_like_image, DownloadAllowlist}; +use crate::crawler::safety::{fetch_stream, looks_like_image, DownloadAllowlist}; use crate::crawler::session::{self, ChapterProbe}; -use crate::storage::Storage; +use crate::storage::{Storage, StorageError}; /// Parse the chapter page DOM and return the page images in `pageN` /// order. Filters out the loader `` and any @@ -186,11 +186,17 @@ where } } -/// Fetch all images for one chapter and persist them atomically. On -/// any error after the first storage put, the DB transaction rolls -/// back so the chapter stays at `page_count = 0` and is retried on the -/// next run. Bytes already written to storage become orphans; a future -/// reaper sweeps them. +/// Fetch one chapter's images and persist them. Each image is +/// streamed straight to storage via `Storage::put_stream` after a +/// short prefix is peeked off the body for content-type sniffing — +/// peak memory per concurrent dispatch is one HTTP chunk plus the +/// sniff prefix, not a full multi-MB image. The per-image size cap +/// (`CRAWLER_MAX_IMAGE_BYTES`) is enforced inside the stream so a +/// server that omits Content-Length still can't exhaust memory. The +/// page rows + `page_count` are then written in one short transaction. +/// On any failure the chapter stays at `page_count = 0` (no partial +/// rows) and the blobs already written are deleted best-effort by +/// [`cleanup_orphans`], so a retry starts clean. #[allow(clippy::too_many_arguments)] pub async fn sync_chapter_content( browser: &chromiumoxide::Browser, @@ -205,6 +211,10 @@ pub async fn sync_chapter_content( allowlist: &DownloadAllowlist, max_image_bytes: usize, tor: Option<&crate::crawler::tor::TorController>, + // Optional live-status sink for the realtime page counter. The daemon + // dispatcher passes the shared handle (the chapter has already been + // registered via `begin_chapter`); the CLI / admin resync pass `None`. + progress: Option<&crate::crawler::status::StatusHandle>, ) -> anyhow::Result { // Skip if already fetched, unless caller explicitly forces. if !force_refetch { @@ -260,56 +270,189 @@ pub async fn sync_chapter_content( // Resolve image URLs against the chapter URL (they may be relative). let base = reqwest::Url::parse(source_url).context("parse chapter URL")?; - // Fetch every image bytes-first into memory before writing - // anything. Lets us bail the whole chapter cleanly if any image - // fails — DB stays at page_count=0, no partial rows persisted. - let mut fetched: Vec<(i32, Vec, &'static str)> = Vec::with_capacity(images.len()); + // Stream each image straight to storage as it's fetched, capping peak + // memory at a single image rather than the whole chapter. Track the + // keys written so they can be rolled back if a later page (or the + // final DB commit) fails — preserving the all-or-nothing guarantee + // without holding a DB transaction open across the network puts + // (which matters once `Storage` is backed by S3). + let total = images.len(); + // Publish the now-known page total so the dashboard shows "0/N". + if let Some(p) = progress { + p.set_chapter_pages(chapter_id, 0, Some(total)); + } + let mut written_keys: Vec = Vec::with_capacity(total); + let mut stored: Vec = Vec::with_capacity(total); for img in &images { - let url = base.join(&img.url).with_context(|| { - format!("join image URL {} onto {source_url}", img.url) - })?; - rate.wait_for(url.as_str()).await?; - let bytes = fetch_bytes_capped( + match download_and_store_page( + storage, http, - url.as_str(), - Some(source_url), + rate, + &base, + source_url, + manga_id, + chapter_id, + img, allowlist, max_image_bytes, ) - .await? - .to_vec(); - // Reject any non-image response: the only valid output of an - // image URL is an image. `infer` returns None on truncated - // bytes too, which also wants to be a failure not a silent - // `.bin` extension. - if !looks_like_image(&bytes) { - anyhow::bail!( - "image URL {url} returned non-image bytes \ - (first 16: {:?}); refusing to store as binary blob", - &bytes.get(..16.min(bytes.len())) - ); + .await + { + Ok(page) => { + written_keys.push(page.storage_key.clone()); + stored.push(page); + // Live page counter: push the climbing count to subscribers. + if let Some(p) = progress { + p.set_chapter_pages(chapter_id, stored.len(), Some(total)); + } + } + Err(e) => { + cleanup_orphans(storage, &written_keys).await; + return Err(e); + } } - let ext = infer::get(&bytes) - .map(|k| k.extension()) - .expect("looks_like_image asserted infer succeeded"); - fetched.push((img.page_number, bytes, ext)); } - // Atomic write: storage puts + page row inserts + page_count - // update, all in one transaction. If anything fails, rollback + - // the chapter is retried next run. Storage orphans the bytes; a - // reaper sweeps them later. - let mut tx = db.begin().await.context("open chapter sync tx")?; - for (page_number, bytes, ext) in &fetched { - let key = format!( - "mangas/{manga_id}/chapters/{chapter_id}/pages/{:04}.{ext}", - page_number + // Short transaction: page rows + page_count only, no network I/O. On + // failure, roll back the stored bytes so the chapter stays at + // page_count=0 and is retried cleanly next run. + if let Err(e) = persist_pages(db, chapter_id, &stored).await { + cleanup_orphans(storage, &written_keys).await; + return Err(e); + } + + Ok(SyncOutcome::Fetched { pages: stored.len() }) +} + +/// A page image that has been written to storage and is awaiting its DB +/// row. Carries everything `persist_pages` needs. +pub(crate) struct StoredPage { + page_number: i32, + storage_key: String, + content_type: String, +} + +/// Bytes accumulated for content-type sniffing. `infer` only needs the +/// first few bytes for image formats (the longest signature in our +/// allow-list is AVIF at 12 bytes), but we read up to this many so a +/// fragmented TCP frame still produces a confident sniff and the +/// "first 16 bytes" diagnostic in the error path is useful. +const SNIFF_PREFIX_BYTES: usize = 64; + +/// Download a single page image, validate it's really an image, and +/// stream it to storage. Returns the storage key + content type. Does +/// not touch the DB — persistence is batched into one short transaction +/// afterward. +/// +/// Streaming path: we peek the first [`SNIFF_PREFIX_BYTES`] from the +/// HTTP body to determine the file extension (and thus the storage +/// key), then re-emit those bytes followed by the rest of the response +/// stream via `Storage::put_stream`. Peak memory per concurrent +/// dispatch is one HTTP chunk (~16 KiB) plus the sniff prefix, not a +/// full multi-MB image. The per-image cap is enforced as bytes flow. +#[allow(clippy::too_many_arguments)] +async fn download_and_store_page( + storage: &dyn Storage, + http: &reqwest::Client, + rate: &HostRateLimiters, + base: &reqwest::Url, + source_url: &str, + manga_id: Uuid, + chapter_id: Uuid, + img: &ChapterImage, + allowlist: &DownloadAllowlist, + max_image_bytes: usize, +) -> anyhow::Result { + use futures_util::StreamExt as _; + let url = base + .join(&img.url) + .with_context(|| format!("join image URL {} onto {source_url}", img.url))?; + rate.wait_for(url.as_str()).await?; + let resp = fetch_stream(http, url.as_str(), Some(source_url), allowlist).await?; + let mut body = resp.bytes_stream(); + + // Drain chunks until we have enough bytes to sniff confidently + // (or the body is shorter than the prefix). Enforces the per-image + // cap on the prefix accumulation too. + let mut prefix = bytes::BytesMut::new(); + while prefix.len() < SNIFF_PREFIX_BYTES { + let Some(chunk) = body.next().await else { break }; + let chunk = chunk + .with_context(|| format!("stream chunk for {url}"))?; + if prefix.len().saturating_add(chunk.len()) > max_image_bytes { + anyhow::bail!( + "image {url} exceeds {max_image_bytes}-byte cap (received >{}+{})", + prefix.len(), + chunk.len() + ); + } + prefix.extend_from_slice(&chunk); + } + let prefix = prefix.freeze(); + + // Reject any non-image response: the only valid output of an image + // URL is an image. `infer` returns None on truncated bytes too, + // which is also a failure not a silent `.bin` extension. + if !looks_like_image(&prefix) { + anyhow::bail!( + "image URL {url} returned non-image bytes \ + (first 16: {:?}); refusing to store as binary blob", + &prefix.get(..16.min(prefix.len())) ); - storage - .put(&key, bytes) - .await - .with_context(|| format!("put {key}"))?; - // (chapter_id, page_number) is unique — re-runs idempotent. + } + let ext = infer::get(&prefix) + .map(|k| k.extension()) + .expect("looks_like_image asserted infer succeeded"); + let key = format!( + "mangas/{manga_id}/chapters/{chapter_id}/pages/{:04}.{ext}", + img.page_number + ); + + // Build a single stream of (prefix + remaining body) and pipe it + // straight to storage. The cap is enforced via a running total in + // the stream adapter so a server that omits Content-Length still + // can't exhaust memory. + let prefix_stream = futures_util::stream::once(async move { + Ok::(prefix) + }); + let prefix_len = SNIFF_PREFIX_BYTES.min(max_image_bytes); + let mut remaining = max_image_bytes.saturating_sub(prefix_len); + let url_for_err = url.clone(); + let rest_stream = body.map(move |frame| match frame { + Ok(chunk) => { + if chunk.len() > remaining { + return Err(StorageError::Io(std::io::Error::other(format!( + "image {url_for_err} exceeds {max_image_bytes}-byte cap" + )))); + } + remaining -= chunk.len(); + Ok(chunk) + } + Err(e) => Err(StorageError::Io(std::io::Error::other(format!( + "stream chunk for {url_for_err}: {e}" + )))), + }); + let combined = prefix_stream.chain(rest_stream); + storage + .put_stream(&key, Box::pin(combined)) + .await + .with_context(|| format!("put_stream {key}"))?; + Ok(StoredPage { + page_number: img.page_number, + storage_key: key, + content_type: format!("image/{ext}"), + }) +} + +/// Persist the page rows + chapter `page_count` in one short transaction. +/// `(chapter_id, page_number)` is unique so re-runs are idempotent. +pub(crate) async fn persist_pages( + db: &PgPool, + chapter_id: Uuid, + stored: &[StoredPage], +) -> anyhow::Result<()> { + let mut tx = db.begin().await.context("open chapter sync tx")?; + for page in stored { sqlx::query( "INSERT INTO pages (chapter_id, page_number, storage_key, content_type) VALUES ($1, $2, $3, $4) @@ -318,22 +461,36 @@ pub async fn sync_chapter_content( content_type = EXCLUDED.content_type", ) .bind(chapter_id) - .bind(page_number) - .bind(&key) - .bind(format!("image/{ext}")) + .bind(page.page_number) + .bind(&page.storage_key) + .bind(&page.content_type) .execute(&mut *tx) .await - .with_context(|| format!("insert page row {page_number}"))?; + .with_context(|| format!("insert page row {}", page.page_number))?; } sqlx::query("UPDATE chapters SET page_count = $1 WHERE id = $2") - .bind(fetched.len() as i32) + .bind(stored.len() as i32) .bind(chapter_id) .execute(&mut *tx) .await .context("update page_count")?; tx.commit().await.context("commit chapter sync")?; + Ok(()) +} - Ok(SyncOutcome::Fetched { pages: fetched.len() }) +/// Best-effort delete of partially-written page blobs after a chapter sync +/// fails, so a retry doesn't accumulate orphans. Errors are logged, not +/// raised — a leftover blob is harmless and a future reaper can sweep it. +pub(crate) async fn cleanup_orphans(storage: &dyn Storage, keys: &[String]) { + for key in keys { + if let Err(e) = storage.delete(key).await { + tracing::warn!( + %key, + error = ?e, + "failed to delete orphaned page blob after chapter sync failure" + ); + } + } } // Suppress unused-import warning for `session::registrable_domain` @@ -347,6 +504,90 @@ fn _keep_session_in_scope() { #[cfg(test)] mod tests { use super::*; + use crate::storage::LocalStorage; + + #[tokio::test] + async fn cleanup_orphans_deletes_written_keys() { + let dir = tempfile::tempdir().unwrap(); + let storage = LocalStorage::new(dir.path()); + let keys = vec![ + "mangas/m/chapters/c/pages/0001.jpg".to_string(), + "mangas/m/chapters/c/pages/0002.jpg".to_string(), + ]; + for k in &keys { + storage.put(k, b"\xff\xd8\xff\xe0 jpeg-ish").await.unwrap(); + assert!(storage.exists(k).await.unwrap()); + } + cleanup_orphans(&storage, &keys).await; + for k in &keys { + assert!(!storage.exists(k).await.unwrap(), "{k} should be deleted"); + } + } + + #[tokio::test] + async fn cleanup_orphans_tolerates_missing_keys() { + // A key that was never written (e.g. the put itself failed) must + // not make cleanup error — it's best-effort. + let dir = tempfile::tempdir().unwrap(); + let storage = LocalStorage::new(dir.path()); + cleanup_orphans(&storage, &["never/written.jpg".to_string()]).await; + } + + #[sqlx::test(migrations = "./migrations")] + async fn persist_pages_inserts_rows_and_sets_page_count(pool: PgPool) { + let manga_id = Uuid::new_v4(); + let chapter_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, 'T')") + .bind(manga_id) + .execute(&pool) + .await + .unwrap(); + sqlx::query("INSERT INTO chapters (id, manga_id, number) VALUES ($1, $2, 1)") + .bind(chapter_id) + .bind(manga_id) + .execute(&pool) + .await + .unwrap(); + + let stored = vec![ + StoredPage { + page_number: 1, + storage_key: "k/0001.jpg".into(), + content_type: "image/jpeg".into(), + }, + StoredPage { + page_number: 2, + storage_key: "k/0002.jpg".into(), + content_type: "image/jpeg".into(), + }, + ]; + persist_pages(&pool, chapter_id, &stored).await.unwrap(); + + let page_count: i32 = + sqlx::query_scalar("SELECT page_count FROM chapters WHERE id = $1") + .bind(chapter_id) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(page_count, 2); + let rows: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM pages WHERE chapter_id = $1") + .bind(chapter_id) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(rows, 2); + + // Idempotent re-run (force refetch path): same rows, page_count stable. + persist_pages(&pool, chapter_id, &stored).await.unwrap(); + let rows2: i64 = + sqlx::query_scalar("SELECT COUNT(*) FROM pages WHERE chapter_id = $1") + .bind(chapter_id) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(rows2, 2, "re-run is idempotent via ON CONFLICT"); + } #[test] fn parse_chapter_pages_skips_loader_and_sorts_by_id() { diff --git a/backend/src/crawler/daemon.rs b/backend/src/crawler/daemon.rs index eb3e512..2c808d9 100644 --- a/backend/src/crawler/daemon.rs +++ b/backend/src/crawler/daemon.rs @@ -48,6 +48,7 @@ use tokio_util::sync::CancellationToken; use crate::crawler::content::SyncOutcome; use crate::crawler::jobs::{self, JobPayload, Lease, KIND_SYNC_CHAPTER_CONTENT}; use crate::crawler::pipeline; +use crate::crawler::status::{Phase, StatusHandle}; /// Fixed `pg_try_advisory_lock` key. ASCII "MANGALRD" interpreted as a /// big-endian i64. Hardcoded so every replica agrees on the lock identity @@ -56,6 +57,15 @@ pub const CRON_LOCK_KEY: i64 = 0x4D414E47414C5244; const STATE_KEY_LAST_TICK: &str = "last_metadata_tick_at"; +/// Lease window handed to `jobs::lease`. Kept short, but continuously +/// extended by the per-job heartbeat (see [`WorkerContext::process_lease`]) +/// so a long-but-healthy job never lapses and gets stolen. +const LEASE_DURATION: Duration = Duration::from_secs(60); + +/// How often the heartbeat renews the lease while a job runs. A third of +/// the lease window leaves two missed-beat's slack before expiry. +const LEASE_HEARTBEAT: Duration = Duration::from_secs(20); + #[async_trait] pub trait MetadataPass: Send + Sync { async fn run(&self) -> anyhow::Result; @@ -77,6 +87,13 @@ pub struct DaemonConfig { pub tz: Tz, pub retention_days: u32, pub session_expired: Arc, + /// Live status surface updated by the cron + workers. + pub status: StatusHandle, + /// Hard upper bound on a single job's dispatch. A job that exceeds it + /// is acked failed (exponential backoff) rather than wedging a worker + /// forever. Must exceed [`LEASE_HEARTBEAT`] and the realistic + /// single-job runtime. + pub job_timeout: Duration, /// Tasks that should run alongside the cron + workers and be cancelled /// on shutdown. Used to hand the daemon ownership of the browser /// manager's idle reaper. @@ -123,6 +140,8 @@ pub fn spawn(pool: PgPool, cancel: CancellationToken, cfg: DaemonConfig) -> Daem tz, retention_days, session_expired, + status, + job_timeout, extra_tasks, } = cfg; @@ -134,6 +153,7 @@ pub fn spawn(pool: PgPool, cancel: CancellationToken, cfg: DaemonConfig) -> Daem tz, retention_days, metadata, + status: status.clone(), }; join.spawn(async move { ctx.run().await }); } else { @@ -146,6 +166,8 @@ pub fn spawn(pool: PgPool, cancel: CancellationToken, cfg: DaemonConfig) -> Daem cancel: cancel.clone(), dispatcher: Arc::clone(&dispatcher), session_expired: Arc::clone(&session_expired), + status: status.clone(), + job_timeout, id: worker_id, }; join.spawn(async move { ctx.run().await }); @@ -169,6 +191,7 @@ struct CronContext { tz: Tz, retention_days: u32, metadata: Arc, + status: StatusHandle, } impl CronContext { @@ -196,6 +219,11 @@ impl CronContext { // (NTP step, suspend/resume) don't strand us on a stale instant. let next = next_fire(Utc::now(), self.daily_at, self.tz); let wait = (next - Utc::now()).to_std().unwrap_or(Duration::ZERO); + self.status + .set_phase(Phase::Idle { + next_fire: Some(next), + }) + .await; tracing::info!( next_fire_utc = %next.to_rfc3339(), wait_seconds = wait.as_secs(), @@ -243,9 +271,13 @@ impl CronContext { let metadata = &self.metadata; let pool = &self.pool; let retention_days = self.retention_days; + let status = &self.status; let body = async move { match metadata.run().await { - Ok(stats) => tracing::info!(?stats, "cron: metadata pass done"), + Ok(stats) => { + status.record_pass(&stats, Utc::now()).await; + tracing::info!(?stats, "cron: metadata pass done"); + } Err(e) => tracing::error!(?e, "cron: metadata pass failed"), } match pipeline::enqueue_bookmarked_pending(pool).await { @@ -283,6 +315,8 @@ struct WorkerContext { cancel: CancellationToken, dispatcher: Arc, session_expired: Arc, + status: StatusHandle, + job_timeout: Duration, id: usize, } @@ -303,7 +337,7 @@ impl WorkerContext { &self.pool, Some(KIND_SYNC_CHAPTER_CONTENT), 1, - Duration::from_secs(60), + LEASE_DURATION, ) .await { @@ -341,9 +375,59 @@ impl WorkerContext { } } - let outcome = AssertUnwindSafe(self.dispatcher.dispatch(lease.payload.clone())) - .catch_unwind() - .await; + // Heartbeat: keep the lease fresh while the (potentially long) + // dispatch runs, so a slow-but-healthy job is never re-leased and + // never inflates `attempts` toward `max_attempts`. Stops itself + // once the job is no longer ours (renew returns false). + let heartbeat = { + let hb_pool = self.pool.clone(); + let hb_id = lease.id; + tokio::spawn(async move { + loop { + tokio::time::sleep(LEASE_HEARTBEAT).await; + match jobs::renew(&hb_pool, hb_id, LEASE_DURATION).await { + Ok(true) => {} + Ok(false) => break, + Err(e) => { + tracing::warn!(lease_id = %hb_id, ?e, "heartbeat renew failed"); + } + } + } + }) + }; + + // The "currently crawling" chapter (with its live page count) is + // registered by the dispatcher itself (RealChapterDispatcher) so it + // carries the manga/chapter identity + page progress and is removed + // via an RAII guard on every exit path. + + // Outer timeout: a dispatch that exceeds `job_timeout` is acked + // failed (exponential backoff) rather than wedging the worker. + let dispatch = AssertUnwindSafe(self.dispatcher.dispatch(lease.payload.clone())) + .catch_unwind(); + let outcome = tokio::time::timeout(self.job_timeout, dispatch).await; + heartbeat.abort(); + + let outcome = match outcome { + Ok(o) => o, + Err(_elapsed) => { + tracing::warn!( + worker = self.id, + lease_id = %lease.id, + timeout_secs = self.job_timeout.as_secs(), + "worker: dispatch timed out — ack failed" + ); + let _ = jobs::ack_failed( + &self.pool, + lease.id, + "dispatch timed out", + lease.attempts, + lease.max_attempts, + ) + .await; + return; + } + }; match outcome { Ok(Ok(SyncOutcome::Fetched { .. } | SyncOutcome::Skipped)) => { let _ = jobs::ack_done(&self.pool, lease.id).await; @@ -355,6 +439,8 @@ impl WorkerContext { "session expired — workers will idle until restart" ); self.session_expired.store(true, Ordering::Release); + // Push the session-expired flip to live status subscribers. + self.status.poke(); let _ = jobs::release(&self.pool, lease.id).await; } Ok(Err(e)) => { diff --git a/backend/src/crawler/jobs.rs b/backend/src/crawler/jobs.rs index ea7eaa9..9f65d1b 100644 --- a/backend/src/crawler/jobs.rs +++ b/backend/src/crawler/jobs.rs @@ -66,16 +66,33 @@ pub struct Lease { pub max_attempts: i32, } -/// Exponential backoff for `ack_failed` retries. `attempts` is the -/// post-increment value reported by `lease()` (so the first failure has -/// `attempts == 1` and waits 60s, the second 120s, etc.). Capped at 1h to -/// avoid runaway long sleeps that would outlive the daemon process. -fn backoff_for(attempts: i32) -> Duration { +/// Deterministic exponential backoff base for `ack_failed` retries. +/// `attempts` is the post-increment value reported by `lease()` (so the +/// first failure has `attempts == 1` and waits 60s, the second 120s, +/// etc.). Capped at 1h to avoid runaway long sleeps that would outlive +/// the daemon process. Jitter is applied separately by [`apply_jitter`]. +fn backoff_base(attempts: i32) -> Duration { let shift = attempts.saturating_sub(1).clamp(0, 20) as u32; let secs = 60u64.saturating_mul(1u64 << shift); Duration::from_secs(secs.min(3600)) } +/// Apply ±20% jitter to a backoff duration. `jitter` is a fraction in +/// `[0.0, 1.0)` (e.g. `rand::random::()`), mapped to a multiplier in +/// `[0.8, 1.2)`. Pure so the bounds stay unit-testable. Spreading retries +/// avoids a thundering herd when a source outage fails many jobs at once. +fn apply_jitter(base: Duration, jitter: f64) -> Duration { + let frac = jitter.clamp(0.0, 1.0); + let mult = 0.8 + 0.4 * frac; // [0.8, 1.2) + Duration::from_secs((base.as_secs_f64() * mult).round() as u64) +} + +/// Jittered exponential backoff for `ack_failed`. Wraps [`backoff_base`] +/// with a random ±20% spread. +fn backoff_for(attempts: i32) -> Duration { + apply_jitter(backoff_base(attempts), rand::random::()) +} + /// Insert a new pending job. For `SyncChapterContent` payloads the /// partial unique index `crawler_jobs_chapter_content_dedup_idx` blocks /// a second `(pending|running)` insert per chapter_id, returning @@ -159,6 +176,35 @@ pub async fn lease( Ok(leases) } +/// Extend the lease on a still-owned `running` job. Returns `true` if the +/// row was updated (we still hold the lease), `false` if the job is no +/// longer `running` (re-leased after a missed heartbeat, or already +/// acked) — the caller's heartbeat loop should stop. The `state = +/// 'running'` guard mirrors [`ack_done`]'s rationale. +/// +/// This is the heartbeat primitive: a worker renews periodically while a +/// long-but-healthy job runs so `leased_until` never lapses, which would +/// otherwise let another worker steal the in-flight job and spuriously +/// inflate `attempts` toward `max_attempts`. +pub async fn renew( + pool: &PgPool, + lease_id: Uuid, + lease_duration: Duration, +) -> sqlx::Result { + let lease_ms: i64 = lease_duration.as_millis().min(i64::MAX as u128) as i64; + let res = sqlx::query( + "UPDATE crawler_jobs \ + SET leased_until = now() + ($2::bigint || ' milliseconds')::interval, \ + updated_at = now() \ + WHERE id = $1 AND state = 'running'", + ) + .bind(lease_id) + .bind(lease_ms) + .execute(pool) + .await?; + Ok(res.rows_affected() > 0) +} + /// Mark a leased job as successfully completed. The `state = 'running'` /// predicate guards against a late ack from a worker whose lease expired /// and was already re-leased by another worker: without it, the late ack @@ -278,19 +324,48 @@ mod tests { use super::*; #[test] - fn backoff_grows_exponentially_and_caps_at_one_hour() { + fn backoff_base_grows_exponentially_and_caps_at_one_hour() { // attempts == 1 → 60s, doubling each step. - assert_eq!(backoff_for(1), Duration::from_secs(60)); - assert_eq!(backoff_for(2), Duration::from_secs(120)); - assert_eq!(backoff_for(3), Duration::from_secs(240)); - assert_eq!(backoff_for(4), Duration::from_secs(480)); - assert_eq!(backoff_for(5), Duration::from_secs(960)); - assert_eq!(backoff_for(6), Duration::from_secs(1920)); + assert_eq!(backoff_base(1), Duration::from_secs(60)); + assert_eq!(backoff_base(2), Duration::from_secs(120)); + assert_eq!(backoff_base(3), Duration::from_secs(240)); + assert_eq!(backoff_base(4), Duration::from_secs(480)); + assert_eq!(backoff_base(5), Duration::from_secs(960)); + assert_eq!(backoff_base(6), Duration::from_secs(1920)); // 7th: 60 * 64 = 3840 → capped to 3600. - assert_eq!(backoff_for(7), Duration::from_secs(3600)); - assert_eq!(backoff_for(20), Duration::from_secs(3600)); + assert_eq!(backoff_base(7), Duration::from_secs(3600)); + assert_eq!(backoff_base(20), Duration::from_secs(3600)); // Garbage / zero / negatives stay sane. - assert_eq!(backoff_for(0), Duration::from_secs(60)); - assert_eq!(backoff_for(-5), Duration::from_secs(60)); + assert_eq!(backoff_base(0), Duration::from_secs(60)); + assert_eq!(backoff_base(-5), Duration::from_secs(60)); + } + + #[test] + fn apply_jitter_stays_within_plus_minus_twenty_percent() { + let base = Duration::from_secs(100); + // Lower bound (jitter = 0.0) → 0.8x. + assert_eq!(apply_jitter(base, 0.0), Duration::from_secs(80)); + // Midpoint (jitter = 0.5) → 1.0x. + assert_eq!(apply_jitter(base, 0.5), Duration::from_secs(100)); + // Upper end (jitter → 1.0) → ~1.2x. + assert_eq!(apply_jitter(base, 1.0), Duration::from_secs(120)); + // Out-of-range inputs are clamped, never panic. + assert_eq!(apply_jitter(base, -3.0), Duration::from_secs(80)); + assert_eq!(apply_jitter(base, 9.0), Duration::from_secs(120)); + } + + #[test] + fn backoff_for_random_jitter_stays_in_band() { + // The production wrapper draws its own randomness; assert the + // result for a mid-range attempt always lands within the jitter + // band of the base, across many draws. + let base = backoff_base(3).as_secs_f64(); // 240s + for _ in 0..1000 { + let v = backoff_for(3).as_secs_f64(); + assert!( + v >= base * 0.8 - 1.0 && v <= base * 1.2 + 1.0, + "jittered backoff {v} outside band of base {base}" + ); + } } } diff --git a/backend/src/crawler/mod.rs b/backend/src/crawler/mod.rs index 650a20a..a3465f6 100644 --- a/backend/src/crawler/mod.rs +++ b/backend/src/crawler/mod.rs @@ -26,6 +26,8 @@ pub mod rate_limit; pub mod resync; pub mod safety; pub mod session; +pub mod session_control; pub mod source; +pub mod status; pub mod tor; pub mod url_utils; diff --git a/backend/src/crawler/pipeline.rs b/backend/src/crawler/pipeline.rs index 0dc5516..fb14fdb 100644 --- a/backend/src/crawler/pipeline.rs +++ b/backend/src/crawler/pipeline.rs @@ -65,6 +65,17 @@ pub(crate) fn should_mark_clean_exit( walked_to_completion || hit_stop_condition } +/// Circuit-breaker: abort the walk once `consecutive` `fetch_manga` +/// failures reach `threshold`. A `threshold` of 0 disables the breaker +/// (unbounded — the legacy behaviour). When it fires the caller must NOT +/// mark a clean exit, so the next tick does a recovery sweep over the +/// catalog tail the aborted pass never reached. +/// +/// Pure so the rule is unit-testable without the walker. +pub(crate) fn should_abort_pass(consecutive: u32, threshold: u32) -> bool { + threshold > 0 && consecutive >= threshold +} + /// Runs the discover → fetch → upsert → cover → chapter-list-diff pipeline /// for the target source. Pure metadata; chapter content is enqueued as /// separate `SyncChapterContent` jobs by the caller after this returns. @@ -103,6 +114,8 @@ pub async fn run_metadata_pass( skip_chapters: bool, allowlist: &DownloadAllowlist, max_image_bytes: usize, + max_consecutive_failures: u32, + status: Option<&crate::crawler::status::StatusHandle>, tor: Option<&crate::crawler::tor::TorController>, ) -> anyhow::Result { let lease = browser_manager @@ -110,6 +123,9 @@ pub async fn run_metadata_pass( .await .context("acquire browser lease for metadata pass")?; let browser_ref: &chromiumoxide::Browser = &lease; + if let Some(s) = status { + s.set_phase(crate::crawler::status::Phase::WalkingList).await; + } let source = { let s = TargetSource::new(start_url.to_string()); @@ -165,6 +181,11 @@ pub async fn run_metadata_pass( let mut walked_to_completion = false; let mut hit_limit = false; let mut hit_stop_condition = false; + // Circuit-breaker state: consecutive fetch_manga failures. A sustained + // run abort (source outage) leaves the pass un-clean → recovery sweep + // next tick. + let mut consecutive_failures = 0u32; + let mut hit_failure_breaker = false; 'outer: loop { let batch = match walker.next_batch(&ctx).await? { @@ -175,6 +196,17 @@ pub async fn run_metadata_pass( } }; for r in batch { + // Cooperative checkpoint: if a coordinated browser restart is + // pending, yield our (long-lived) lease so the drain can + // proceed instead of stalling for the rest of the walk. The + // pass exits un-clean, so the next tick recovery-sweeps the + // tail we didn't reach. + if browser_manager.is_restart_pending() { + tracing::info!( + "metadata pass: browser restart pending — yielding (recovery sweep next tick)" + ); + break 'outer; + } if max_refs.map(|m| stats.discovered >= m).unwrap_or(false) { hit_limit = true; tracing::info!(cap = ?max_refs, "max_results reached; halting walk"); @@ -198,13 +230,24 @@ pub async fn run_metadata_pass( continue; } stats.discovered += 1; + if let Some(s) = status { + s.set_phase(crate::crawler::status::Phase::FetchingMetadata { + index: stats.discovered, + total: max_refs, + title: r.title.clone(), + }) + .await; + } tracing::info!( idx = stats.discovered, key = %r.source_manga_key, "fetching metadata" ); let manga = match source.fetch_manga(&ctx, &r).await { - Ok(m) => m, + Ok(m) => { + consecutive_failures = 0; + m + } Err(e) => { tracing::warn!( key = %r.source_manga_key, @@ -213,6 +256,17 @@ pub async fn run_metadata_pass( "fetch_manga failed" ); stats.mangas_failed += 1; + consecutive_failures += 1; + if should_abort_pass(consecutive_failures, max_consecutive_failures) { + hit_failure_breaker = true; + tracing::error!( + consecutive_failures, + threshold = max_consecutive_failures, + "metadata pass: too many consecutive fetch_manga failures; \ + aborting (recovery sweep on next tick)" + ); + break 'outer; + } continue; } }; @@ -295,7 +349,16 @@ pub async fn run_metadata_pass( || matches!(upsert.status, repo::crawler::UpsertStatus::Updated); if needs_cover { if let Some(cover_url) = manga.cover_url.as_deref() { - match download_and_store_cover( + // RAII: the guard clears `current_cover` on every + // exit path (success, panic, future early-return). + // Mirrors the chapter-side ChapterGuard. + let _cover_guard = status.map(|s| { + s.begin_cover(crate::crawler::status::CoverTarget { + manga_id: upsert.manga_id, + manga_title: manga.title.clone(), + }) + }); + let cover_result = download_and_store_cover( db, storage, http, @@ -306,8 +369,8 @@ pub async fn run_metadata_pass( allowlist, max_image_bytes, ) - .await - { + .await; + match cover_result { Ok(()) => stats.covers_fetched += 1, Err(e) => tracing::warn!( manga_id = %upsert.manga_id, @@ -390,6 +453,7 @@ pub async fn run_metadata_pass( walked_to_completion, hit_limit, hit_stop_condition, + hit_failure_breaker, exited_cleanly, "metadata pass complete" ); @@ -560,6 +624,7 @@ pub async fn backfill_missing_covers( max_mangas: usize, allowlist: &DownloadAllowlist, max_image_bytes: usize, + status: Option<&crate::crawler::status::StatusHandle>, tor: Option<&crate::crawler::tor::TorController>, ) -> anyhow::Result { let mut stats = CoverBackfillStats::default(); @@ -582,8 +647,13 @@ pub async fn backfill_missing_covers( let browser_ref: &chromiumoxide::Browser = &lease; let ctx = FetchContext { browser: browser_ref, rate, tor }; - for entry in entries { + let total = entries.len(); + for (index, entry) in entries.into_iter().enumerate() { stats.considered += 1; + if let Some(s) = status { + s.set_phase(crate::crawler::status::Phase::CoverBackfill { index, total }) + .await; + } // Metadata-only TargetSource: skip chapter-list parsing so a // missing-cover refetch doesn't soft-drop chapters on a partial // render. Cover URL alone is what we need. @@ -593,8 +663,8 @@ pub async fn backfill_missing_covers( title: String::new(), url: entry.source_url.clone(), }; - let cover_url = match source.fetch_manga(&ctx, &r).await { - Ok(manga) => manga.cover_url, + let manga = match source.fetch_manga(&ctx, &r).await { + Ok(manga) => manga, Err(e) => { tracing::warn!( manga_id = %entry.manga_id, @@ -606,7 +676,7 @@ pub async fn backfill_missing_covers( continue; } }; - let Some(cover_url) = cover_url else { + let Some(cover_url) = manga.cover_url.clone() else { tracing::warn!( manga_id = %entry.manga_id, url = %entry.source_url, @@ -615,7 +685,16 @@ pub async fn backfill_missing_covers( stats.failed += 1; continue; }; - match download_and_store_cover( + // RAII guard: clears the live current_cover on every exit path, + // including a panic inside download_and_store_cover. Mirrors the + // chapter-side ChapterGuard. + let _cover_guard = status.map(|s| { + s.begin_cover(crate::crawler::status::CoverTarget { + manga_id: entry.manga_id, + manga_title: manga.title.clone(), + }) + }); + let cover_result = download_and_store_cover( db, storage, http, @@ -626,8 +705,8 @@ pub async fn backfill_missing_covers( allowlist, max_image_bytes, ) - .await - { + .await; + match cover_result { Ok(()) => stats.fetched += 1, Err(e) => { tracing::warn!( @@ -756,6 +835,18 @@ mod tests { assert!(!should_stop(false, UpsertStatus::New, None)); } + #[test] + fn abort_pass_fires_at_threshold_and_respects_disable() { + // Disabled (0) never fires, no matter how many failures. + assert!(!should_abort_pass(0, 0)); + assert!(!should_abort_pass(100, 0)); + // Below threshold: keep going. + assert!(!should_abort_pass(9, 10)); + // At/above threshold: abort. + assert!(should_abort_pass(10, 10)); + assert!(should_abort_pass(11, 10)); + } + #[test] fn clean_exit_when_walked_to_completion() { // End-of-walk reached the catalog tail — the recovery flag may diff --git a/backend/src/crawler/resync.rs b/backend/src/crawler/resync.rs index 1e77333..f964c93 100644 --- a/backend/src/crawler/resync.rs +++ b/backend/src/crawler/resync.rs @@ -235,7 +235,7 @@ impl ResyncService for RealResyncService { let row = repo::chapter::dispatch_target(&self.db, chapter_id) .await .context("look up chapter_sources for resync")?; - let Some((manga_id, source_url)) = row else { + let Some((manga_id, source_url, _title, _number)) = row else { return Err(ResyncError::NoChapterSource.into()); }; @@ -257,6 +257,8 @@ impl ResyncService for RealResyncService { &self.download_allowlist, self.max_image_bytes, self.tor.as_deref(), + // Admin resync isn't a daemon worker slot — no live status. + None, ) .await; drop(lease); diff --git a/backend/src/crawler/safety.rs b/backend/src/crawler/safety.rs index deb7f61..3ce6c71 100644 --- a/backend/src/crawler/safety.rs +++ b/backend/src/crawler/safety.rs @@ -241,6 +241,31 @@ pub async fn fetch_bytes_capped( .with_context(|| format!("download body for {url}")) } +/// Send `req` and return the response body as a stream after the +/// safety check + 2xx status check. Caller owns chunking, capping, and +/// piping to storage. Used by `download_and_store_page` so peak memory +/// stays at one chunk per concurrent dispatch instead of one full +/// image. +pub async fn fetch_stream( + http: &reqwest::Client, + url: &str, + referer: Option<&str>, + allow: &DownloadAllowlist, +) -> anyhow::Result { + is_safe_url(url, allow).with_context(|| format!("reject unsafe URL {url}"))?; + let mut req = http.get(url); + if let Some(r) = referer { + req = req.header(reqwest::header::REFERER, r); + } + let resp = req + .send() + .await + .with_context(|| format!("GET {url}"))? + .error_for_status() + .with_context(|| format!("non-2xx for {url}"))?; + Ok(resp) +} + /// True when `bytes` sniffs as one of the *renderable* image formats /// the `/files/*key` endpoint can serve with a correct Content-Type: /// JPEG, PNG, WebP, GIF, AVIF. Matches the upload pipeline's diff --git a/backend/src/crawler/session_control.rs b/backend/src/crawler/session_control.rs new file mode 100644 index 0000000..c9e82d2 --- /dev/null +++ b/backend/src/crawler/session_control.rs @@ -0,0 +1,184 @@ +//! Runtime-updatable crawler session (PHPSESSID). +//! +//! At startup the session comes from `CRAWLER_PHPSESSID`, but it expires +//! and previously needed a container restart to refresh. This controller +//! lets an admin push a fresh cookie at runtime: it rewrites the reqwest +//! cookie jar (CDN image fetches), updates the in-memory value the browser +//! `on_launch` hook reads, persists it to `crawler_state` (so it survives +//! a restart), and clears the sticky `session_expired` flag. A subsequent +//! coordinated browser restart re-runs `on_launch`, re-injecting the new +//! cookie into Chromium and re-probing. + +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::Arc; + +use anyhow::Context; +use sqlx::PgPool; +use tokio::sync::RwLock; + +use crate::repo; + +pub struct SessionController { + /// Current PHPSESSID — what `on_launch` injects into a fresh browser. + phpsessid: RwLock>, + /// The same `Arc` handed to the reqwest client; updating it here + /// updates the client's cookies (the jar is internally mutable). + cookie_jar: Arc, + cookie_domain: Option, + start_url: Option, + db: PgPool, + session_expired: Arc, +} + +impl SessionController { + pub fn new( + initial: Option, + cookie_jar: Arc, + cookie_domain: Option, + start_url: Option, + db: PgPool, + session_expired: Arc, + ) -> Arc { + Arc::new(Self { + phpsessid: RwLock::new(initial), + cookie_jar, + cookie_domain, + start_url, + db, + session_expired, + }) + } + + /// The PHPSESSID a fresh browser should inject (None when unset). + pub async fn current(&self) -> Option { + self.phpsessid.read().await.clone() + } + + /// Whether the sticky session-expired flag is set (chapter workers + /// idle while true). + pub fn is_expired(&self) -> bool { + self.session_expired.load(Ordering::Acquire) + } + + /// Clear the session-expired flag without changing the cookie — used + /// when the operator knows the session is fine and wants workers to + /// resume immediately. + pub fn clear_expired(&self) { + self.session_expired.store(false, Ordering::Release); + } + + /// Update the session everywhere: reqwest jar, in-memory value, and + /// persisted `crawler_state`. Clears the session-expired flag. Does + /// NOT relaunch the browser — the caller triggers a coordinated + /// restart so `on_launch` re-injects + re-probes. + pub async fn update(&self, sid: &str) -> anyhow::Result<()> { + let sid = sid.trim().to_string(); + anyhow::ensure!(!sid.is_empty(), "PHPSESSID must not be empty"); + // The value is spliced into a cookie string and a CDP CookieParam. + // PHPSESSID values produced by PHP are URL-safe base64 alphanumerics + // plus a small set of punctuation depending on session.sid_bits_per_ + // character. An allow-list (rather than a blocklist of control chars + // + `;,`) makes the check robust against future cookie syntax + // extensions and forces a paste that includes whitespace, quotes, + // backslashes, etc. — typical signs of a botched copy-paste — to + // be rejected early. + anyhow::ensure!( + sid.chars().all(is_phpsessid_char), + "PHPSESSID contains invalid characters" + ); + + if let (Some(domain), Some(start_url)) = (&self.cookie_domain, &self.start_url) { + let cookie_str = format!("PHPSESSID={sid}; Domain={domain}; Path=/"); + let seed_url = + reqwest::Url::parse(start_url).context("parse start_url for cookie seed")?; + self.cookie_jar.add_cookie_str(&cookie_str, &seed_url); + } + *self.phpsessid.write().await = Some(sid.clone()); + repo::crawler::runtime_session_persist(&self.db, &sid) + .await + .context("persist runtime session")?; + self.session_expired.store(false, Ordering::Release); + tracing::info!("crawler session updated at runtime"); + Ok(()) + } + + /// Read a persisted runtime session (if any) from `crawler_state`. + /// Called at startup so a mid-day refresh survives a restart. + pub async fn load_persisted(db: &PgPool) -> Option { + repo::crawler::runtime_session_load(db).await.ok().flatten() + } +} + +/// Characters allowed in a PHPSESSID. PHP's session.sid_bits_per_character +/// produces alphanumerics plus `-` and `,` in the lowest-bit mode, but our +/// audit rejects `,` (cookie delimiter) — operators paste from a browser +/// devtools snapshot, which never embeds raw commas in the SID itself. +/// Underscore is allowed because some sources customise their session +/// alphabet. +fn is_phpsessid_char(c: char) -> bool { + c.is_ascii_alphanumeric() || matches!(c, '-' | '_') +} + +#[cfg(test)] +mod tests { + use super::*; + + fn controller(db: PgPool) -> Arc { + SessionController::new( + None, + Arc::new(reqwest::cookie::Jar::default()), + Some("example.com".into()), + Some("https://example.com/".into()), + db, + Arc::new(AtomicBool::new(true)), + ) + } + + #[sqlx::test(migrations = "./migrations")] + async fn update_rejects_empty_and_control_chars(pool: PgPool) { + let c = controller(pool); + assert!(c.update(" ").await.is_err(), "empty rejected"); + assert!(c.update("abc\r\ndef").await.is_err(), "CRLF rejected"); + assert!(c.update("ab;Domain=evil").await.is_err(), "semicolon rejected"); + assert!(c.update("x,y").await.is_err(), "comma rejected"); + } + + #[sqlx::test(migrations = "./migrations")] + async fn update_rejects_non_alphanumeric_pastes(pool: PgPool) { + // Allow-list tightening (M6): pastes that include whitespace, + // quotes, slashes, backslashes, `=`, etc. are typical signs of a + // botched copy-paste and must be rejected outright. + let c = controller(pool); + for bad in ["ab cd", "ab\"cd", "ab=cd", "ab/cd", "ab\\cd", "ab+cd", "ab.cd"] { + assert!(c.update(bad).await.is_err(), "{bad:?} should be rejected"); + } + // Allowed cases (sanity): plain alphanumerics, '-' and '_'. + assert!(c.update("abc_DEF-123").await.is_ok()); + } + + #[sqlx::test(migrations = "./migrations")] + async fn update_persists_and_clears_expired_then_round_trips(pool: PgPool) { + let c = controller(pool.clone()); + c.update("good-sid-123").await.unwrap(); + assert_eq!(c.current().await.as_deref(), Some("good-sid-123")); + assert!(!c.is_expired(), "update clears the expired flag"); + // Persisted to crawler_state and readable by a fresh load. + assert_eq!( + SessionController::load_persisted(&pool).await.as_deref(), + Some("good-sid-123") + ); + } + + #[sqlx::test(migrations = "./migrations")] + async fn clear_expired_flips_sticky_flag_without_touching_session(pool: PgPool) { + // The flag starts `true` per `controller(pool)`'s test wiring. + let c = controller(pool); + assert!(c.is_expired(), "test fixture starts with the flag set"); + c.clear_expired(); + assert!(!c.is_expired(), "clear_expired flips the sticky flag to false"); + assert!( + c.current().await.is_none(), + "clear_expired does not invent a session" + ); + } +} diff --git a/backend/src/crawler/status.rs b/backend/src/crawler/status.rs new file mode 100644 index 0000000..5875822 --- /dev/null +++ b/backend/src/crawler/status.rs @@ -0,0 +1,456 @@ +//! Live, in-process crawler status. +//! +//! The metadata pass runs inline in the cron tick (it is not a +//! `crawler_jobs` row), so without this surface "what is the crawler doing +//! right now" is unanswerable from the dashboard. The daemon publishes its +//! current [`Phase`], the chapters being crawled right now (with a live +//! page count), and the cover being fetched into a shared [`StatusHandle`]; +//! the admin endpoint reads a [`CrawlerStatus`] snapshot and composes it +//! with DB-derived counts + the session/browser flags. +//! +//! NOTE: this is per-process state. The deployment is a single server +//! (see CLAUDE.md), so an in-memory handle is sufficient; durable signals +//! (last-pass summary, runtime session) are persisted in `crawler_state`. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use chrono::{DateTime, Utc}; +use serde::Serialize; +use tokio::sync::{watch, RwLock}; +use uuid::Uuid; + +use crate::crawler::pipeline::MetadataStats; + +/// What the daemon's metadata pass is doing right now. Serialised with an +/// internal `state` tag so the frontend can switch on it. +#[derive(Clone, Debug, Serialize)] +#[serde(tag = "state", rename_all = "snake_case")] +pub enum Phase { + /// Sleeping until the next scheduled metadata pass. + Idle { next_fire: Option> }, + /// Walking the source catalog list pages. + WalkingList, + /// Fetching one manga's metadata. `index`/`total` drive a progress bar + /// (`total` is `None` when the source size is unknown / uncapped). + FetchingMetadata { + index: usize, + total: Option, + title: String, + }, + /// Backfilling covers that failed on first attempt. `index`/`total` + /// track progress through this tick's batch. + CoverBackfill { index: usize, total: usize }, +} + +/// A chapter being downloaded right now, with a live page count. Keyed in +/// the status by `chapter_id`; inserted by the dispatcher when a job starts +/// and removed (via an RAII guard) when it finishes, panics, or times out. +#[derive(Clone, Debug, Serialize)] +pub struct ActiveChapter { + pub manga_id: Uuid, + pub manga_title: String, + pub chapter_id: Uuid, + pub chapter_number: i32, + pub pages_done: usize, + /// `None` until the chapter page list has been parsed. + pub pages_total: Option, +} + +/// The manga whose cover is being downloaded right now. +#[derive(Clone, Debug, Serialize)] +pub struct CoverTarget { + pub manga_id: Uuid, + pub manga_title: String, +} + +/// Summary of the most recent metadata pass (persisted across restarts in +/// `crawler_state` by the cron; mirrored here for the live read). +#[derive(Clone, Debug, Serialize, Default)] +pub struct LastPass { + pub at: Option>, + pub discovered: usize, + pub upserted: usize, + pub covers_fetched: usize, + pub mangas_failed: usize, +} + +/// A point-in-time snapshot returned by [`StatusHandle::snapshot`]. The +/// session/browser/queue fields are composed at read time by the endpoint +/// (they live elsewhere), so they are not stored here. +#[derive(Clone, Debug, Serialize)] +pub struct CrawlerStatus { + pub phase: Phase, + /// Number of configured chapter workers (for "N busy / M workers"). + pub worker_count: usize, + /// Chapters being downloaded right now, with live page counts. + pub active_chapters: Vec, + pub last_pass: LastPass, + /// The cover being downloaded right now, if any. + pub current_cover: Option, +} + +/// Scalar status state held under the async `RwLock`. Active chapters and +/// the current cover live in separate sync maps so per-page updates and +/// RAII removal don't need to `.await` (removal happens in `Drop`). +#[derive(Clone, Debug)] +struct Scalar { + phase: Phase, + worker_count: usize, + last_pass: LastPass, +} + +/// Cloneable handle the daemon tasks use to publish status. Cheap to clone +/// (`Arc`). All writers funnel through the helper methods so locking stays +/// localised. Every mutation bumps a `watch` version so SSE subscribers +/// get pushed an update instead of polling. +#[derive(Clone)] +pub struct StatusHandle { + scalar: Arc>, + /// Currently-downloading chapters keyed by `chapter_id`. A sync mutex so + /// the RAII [`ChapterGuard`]'s `Drop` can remove without `.await`. + active: Arc>>, + /// The cover being downloaded right now (if any). Sync mutex so the + /// RAII [`CoverGuard`]'s `Drop` can clear without `.await`, which is + /// what makes the cleared-on-panic guarantee hold. + current_cover: Arc>>, + /// Monotonic version bumped on every change. SSE handlers `subscribe()` + /// and `await .changed()` for instant pushes; `watch` has no + /// lost-wakeup so a change between snapshots is never missed. + version: Arc>, +} + +/// Lock the active map, recovering from a poisoned mutex. The map values +/// are plain structs and we never hold the lock across a panic-prone +/// section, so resuming on poison is safe — but log it so a real poison +/// (which signals a panic-in-critical-section bug somewhere) doesn't pass +/// in silence. +fn lock_active( + m: &Mutex>, +) -> std::sync::MutexGuard<'_, HashMap> { + m.lock().unwrap_or_else(|e| { + tracing::warn!( + "status::lock_active recovered from a poisoned mutex — \ + this implies a panic somewhere holding the lock" + ); + e.into_inner() + }) +} + +/// Same shape as [`lock_active`] but for the single-slot cover mutex. +fn lock_cover( + m: &Mutex>, +) -> std::sync::MutexGuard<'_, Option> { + m.lock().unwrap_or_else(|e| { + tracing::warn!( + "status::lock_cover recovered from a poisoned mutex — \ + this implies a panic somewhere holding the lock" + ); + e.into_inner() + }) +} + +impl StatusHandle { + pub fn new(num_workers: usize) -> Self { + let (version, _rx) = watch::channel(0u64); + Self { + scalar: Arc::new(RwLock::new(Scalar { + phase: Phase::Idle { next_fire: None }, + worker_count: num_workers.max(1), + last_pass: LastPass::default(), + })), + active: Arc::new(Mutex::new(HashMap::new())), + current_cover: Arc::new(Mutex::new(None)), + version: Arc::new(version), + } + } + + fn bump(&self) { + self.version.send_modify(|v| *v = v.wrapping_add(1)); + } + + /// A receiver whose `.changed()` resolves on the next status change. + pub fn subscribe(&self) -> watch::Receiver { + self.version.subscribe() + } + + /// Signal a change without mutating in-memory state — used when an + /// *external* signal the live snapshot reflects (browser phase, + /// session-expired flag, queue counts) has changed, so subscribers + /// recompose promptly. + pub fn poke(&self) { + self.bump(); + } + + pub async fn set_phase(&self, phase: Phase) { + self.scalar.write().await.phase = phase; + self.bump(); + } + + /// Register a cover-fetch as in flight; returns a guard that clears + /// the current cover when dropped (on completion, panic-unwind, or + /// any future early-return). Last-writer-wins: a guard only clears + /// the slot when it still holds the cover it set (so overlapping + /// guards — not used today, but defensive — don't clobber each + /// other). + pub fn begin_cover(&self, target: CoverTarget) -> CoverGuard { + let manga_id = target.manga_id; + *lock_cover(&self.current_cover) = Some(target); + self.bump(); + CoverGuard { + current_cover: Arc::clone(&self.current_cover), + version: Arc::clone(&self.version), + manga_id, + } + } + + /// Register a chapter as crawling now; returns a guard that removes it + /// when dropped (on completion, panic-unwind, or timeout-drop). + pub fn begin_chapter(&self, chapter: ActiveChapter) -> ChapterGuard { + let id = chapter.chapter_id; + lock_active(&self.active).insert(id, chapter); + self.bump(); + ChapterGuard { + active: Arc::clone(&self.active), + version: Arc::clone(&self.version), + chapter_id: id, + } + } + + /// Update the live page count of an in-flight chapter. Sync (no + /// `.await`) so it's cheap to call once per stored page. + pub fn set_chapter_pages(&self, chapter_id: Uuid, done: usize, total: Option) { + { + let mut map = lock_active(&self.active); + if let Some(c) = map.get_mut(&chapter_id) { + c.pages_done = done; + c.pages_total = total; + } + } + self.bump(); + } + + /// Record a finished metadata pass. Stamps `at` with `now`. + pub async fn record_pass(&self, stats: &MetadataStats, at: DateTime) { + self.scalar.write().await.last_pass = LastPass { + at: Some(at), + discovered: stats.discovered, + upserted: stats.upserted, + covers_fetched: stats.covers_fetched, + mangas_failed: stats.mangas_failed, + }; + self.bump(); + } + + /// Seed the last-pass summary from a persisted `crawler_state` value on + /// startup so the dashboard isn't blank until the first tick. + pub async fn set_last_pass(&self, last: LastPass) { + self.scalar.write().await.last_pass = last; + self.bump(); + } + + pub async fn snapshot(&self) -> CrawlerStatus { + let scalar = self.scalar.read().await.clone(); + let mut active_chapters: Vec = + lock_active(&self.active).values().cloned().collect(); + // Stable, readable order: by chapter number then id. + active_chapters.sort_by(|a, b| { + a.chapter_number + .cmp(&b.chapter_number) + .then(a.chapter_id.cmp(&b.chapter_id)) + }); + let current_cover = lock_cover(&self.current_cover).clone(); + CrawlerStatus { + phase: scalar.phase, + worker_count: scalar.worker_count, + active_chapters, + last_pass: scalar.last_pass, + current_cover, + } + } +} + +/// RAII handle clearing the [`CoverTarget`] from the live status when the +/// cover-fetch finishes, panics, or is dropped on any early-return. +pub struct CoverGuard { + current_cover: Arc>>, + version: Arc>, + /// Manga id whose cover this guard registered. The drop only clears + /// the slot when the stored value still matches — defends against a + /// hypothetical newer guard clobbering this one's clear. + manga_id: Uuid, +} + +impl Drop for CoverGuard { + fn drop(&mut self) { + let mut slot = lock_cover(&self.current_cover); + if slot.as_ref().map(|c| c.manga_id) == Some(self.manga_id) { + *slot = None; + } + self.version.send_modify(|v| *v = v.wrapping_add(1)); + } +} + +/// RAII handle removing an [`ActiveChapter`] from the live status when the +/// chapter dispatch finishes, panics, or is dropped on timeout. +pub struct ChapterGuard { + active: Arc>>, + version: Arc>, + chapter_id: Uuid, +} + +impl Drop for ChapterGuard { + fn drop(&mut self) { + lock_active(&self.active).remove(&self.chapter_id); + self.version.send_modify(|v| *v = v.wrapping_add(1)); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample_chapter(n: i32) -> ActiveChapter { + ActiveChapter { + manga_id: Uuid::new_v4(), + manga_title: "M".into(), + chapter_id: Uuid::new_v4(), + chapter_number: n, + pages_done: 0, + pages_total: None, + } + } + + #[tokio::test] + async fn begin_chapter_shows_in_snapshot_and_guard_removes_on_drop() { + let h = StatusHandle::new(2); + let chap = sample_chapter(7); + let cid = chap.chapter_id; + { + let _guard = h.begin_chapter(chap); + let snap = h.snapshot().await; + assert_eq!(snap.active_chapters.len(), 1); + assert_eq!(snap.active_chapters[0].chapter_id, cid); + assert_eq!(snap.worker_count, 2); + } + // Guard dropped → entry removed. + let snap = h.snapshot().await; + assert!(snap.active_chapters.is_empty()); + } + + #[tokio::test] + async fn set_chapter_pages_updates_live_count() { + let h = StatusHandle::new(1); + let chap = sample_chapter(1); + let cid = chap.chapter_id; + let _guard = h.begin_chapter(chap); + h.set_chapter_pages(cid, 3, Some(20)); + let snap = h.snapshot().await; + assert_eq!(snap.active_chapters[0].pages_done, 3); + assert_eq!(snap.active_chapters[0].pages_total, Some(20)); + // Updating an unknown chapter is a no-op, not a panic. + h.set_chapter_pages(Uuid::new_v4(), 9, Some(9)); + } + + #[tokio::test] + async fn snapshot_sorts_active_chapters_by_number() { + let h = StatusHandle::new(2); + let _g1 = h.begin_chapter(sample_chapter(5)); + let _g2 = h.begin_chapter(sample_chapter(2)); + let snap = h.snapshot().await; + assert_eq!(snap.active_chapters[0].chapter_number, 2); + assert_eq!(snap.active_chapters[1].chapter_number, 5); + } + + #[tokio::test] + async fn cover_guard_sets_then_clears_on_drop() { + let h = StatusHandle::new(1); + let mid = Uuid::new_v4(); + { + let _g = h.begin_cover(CoverTarget { + manga_id: mid, + manga_title: "One Piece".into(), + }); + assert_eq!( + h.snapshot().await.current_cover.map(|c| c.manga_id), + Some(mid) + ); + } + assert!(h.snapshot().await.current_cover.is_none()); + } + + #[tokio::test] + async fn cover_guard_clears_on_panic_drop() { + // Simulate a download_and_store_cover panic: the guard is on the + // stack, the panic unwinds, and the slot must still be cleared. + let h = StatusHandle::new(1); + let mid = Uuid::new_v4(); + let h2 = h.clone(); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _g = h2.begin_cover(CoverTarget { + manga_id: mid, + manga_title: "K-On!".into(), + }); + panic!("simulated cover-download panic"); + })); + assert!(result.is_err()); + assert!(h.snapshot().await.current_cover.is_none()); + } + + #[tokio::test] + async fn cover_guard_does_not_clobber_a_newer_target() { + // Defensive: if a *newer* begin_cover ran before the older + // guard's drop fires, the drop must not clear the newer + // target. (No code today produces overlapping guards, but the + // invariant prevents a future caller from quietly breaking the + // live-cover surface.) + let h = StatusHandle::new(1); + let older = Uuid::new_v4(); + let newer = Uuid::new_v4(); + let g_old = h.begin_cover(CoverTarget { + manga_id: older, + manga_title: "old".into(), + }); + let _g_new = h.begin_cover(CoverTarget { + manga_id: newer, + manga_title: "new".into(), + }); + drop(g_old); + assert_eq!( + h.snapshot().await.current_cover.map(|c| c.manga_id), + Some(newer) + ); + } + + #[tokio::test] + async fn record_pass_captures_stats_and_timestamp() { + let h = StatusHandle::new(1); + let stats = MetadataStats { + discovered: 5, + upserted: 3, + covers_fetched: 2, + mangas_failed: 1, + }; + let at = Utc::now(); + h.record_pass(&stats, at).await; + let snap = h.snapshot().await; + assert_eq!(snap.last_pass.discovered, 5); + assert_eq!(snap.last_pass.upserted, 3); + assert_eq!(snap.last_pass.at, Some(at)); + } + + #[tokio::test] + async fn subscribe_resolves_on_mutation_poke_and_chapter_change() { + let h = StatusHandle::new(1); + let mut rx = h.subscribe(); + h.set_phase(Phase::WalkingList).await; + rx.changed().await.unwrap(); + h.poke(); + rx.changed().await.unwrap(); + // begin_chapter + guard drop each bump the version. + let g = h.begin_chapter(sample_chapter(1)); + rx.changed().await.unwrap(); + drop(g); + rx.changed().await.unwrap(); + } +} diff --git a/backend/src/repo/chapter.rs b/backend/src/repo/chapter.rs index 7bd3fe7..cc1065a 100644 --- a/backend/src/repo/chapter.rs +++ b/backend/src/repo/chapter.rs @@ -138,14 +138,18 @@ pub async fn page_count(pool: &PgPool, id: Uuid) -> sqlx::Result> { /// filter — this resolver stays in lockstep so a chapter that was /// dropped between enqueue and lease isn't dispatched against a stale /// URL. +/// Returns `(manga_id, source_url, manga_title, chapter_number)`. The +/// title + number feed the live "currently crawling" status; the rest is +/// what the dispatcher needs to do the work. pub async fn dispatch_target( pool: &PgPool, chapter_id: Uuid, -) -> sqlx::Result> { +) -> sqlx::Result> { sqlx::query_as( - "SELECT c.manga_id, cs.source_url \ + "SELECT c.manga_id, cs.source_url, m.title, c.number \ FROM chapters c \ JOIN chapter_sources cs ON cs.chapter_id = c.id \ + JOIN mangas m ON m.id = c.manga_id \ WHERE c.id = $1 \ AND cs.dropped_at IS NULL \ ORDER BY cs.last_seen_at DESC \ diff --git a/backend/src/repo/crawler.rs b/backend/src/repo/crawler.rs index 2d8dbe6..915fc30 100644 --- a/backend/src/repo/crawler.rs +++ b/backend/src/repo/crawler.rs @@ -17,8 +17,9 @@ //! Each public function is a transaction boundary so a partial failure //! mid-call leaves the DB in its pre-call state. -use chrono::Utc; -use sqlx::{PgPool, Postgres, Transaction}; +use chrono::{DateTime, Utc}; +use serde::Serialize; +use sqlx::{FromRow, PgPool, Postgres, Transaction}; use uuid::Uuid; use crate::crawler::source::{SourceChapterRef, SourceManga}; @@ -618,3 +619,424 @@ pub async fn last_run_completed_cleanly( .unwrap_or(true)) } +// --------------------------------------------------------------------------- +// Dead-letter jobs: admin observability + requeue. +// --------------------------------------------------------------------------- + +/// A `dead` crawler job joined to its chapter/manga context for the admin +/// dead-letter view. Chapter columns are `Option` because the join is +/// best-effort (the chapter may have been removed since the job died, or +/// the job may be a non-chapter kind). +#[derive(Debug, Clone, Serialize, FromRow)] +pub struct DeadJob { + pub id: Uuid, + pub kind: String, + pub chapter_id: Option, + pub manga_id: Option, + pub manga_title: Option, + pub chapter_number: Option, + pub attempts: i32, + pub max_attempts: i32, + pub last_error: Option, + pub updated_at: DateTime, +} + +/// Paginated list of `dead` jobs, newest-failed first, joined to chapter + +/// manga context. `search` filters on manga title (case-insensitive +/// substring). Returns the page slice plus the unfiltered-by-page total. +pub async fn list_dead_jobs( + pool: &PgPool, + search: Option<&str>, + limit: i64, + offset: i64, +) -> sqlx::Result<(Vec, i64)> { + let search_pat = search + .map(|s| format!("%{}%", s.trim())) + .filter(|p| p.len() > 2); + + let items: Vec = sqlx::query_as( + r#" + SELECT + cj.id, + cj.payload->>'kind' AS kind, + (cj.payload->>'chapter_id')::uuid AS chapter_id, + c.manga_id AS manga_id, + m.title AS manga_title, + c.number AS chapter_number, + cj.attempts, + cj.max_attempts, + cj.last_error, + cj.updated_at + FROM crawler_jobs cj + LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid + LEFT JOIN mangas m ON m.id = c.manga_id + WHERE cj.state = 'dead' + AND ($1::text IS NULL OR m.title ILIKE $1) + ORDER BY cj.updated_at DESC + LIMIT $2 OFFSET $3 + "#, + ) + .bind(&search_pat) + .bind(limit) + .bind(offset) + .fetch_all(pool) + .await?; + + let total: i64 = sqlx::query_scalar( + r#" + SELECT COUNT(*) + FROM crawler_jobs cj + LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid + LEFT JOIN mangas m ON m.id = c.manga_id + WHERE cj.state = 'dead' + AND ($1::text IS NULL OR m.title ILIKE $1) + "#, + ) + .bind(&search_pat) + .fetch_one(pool) + .await?; + + Ok((items, total)) +} + +/// An in-flight chapter-content job (`pending` or `running`) joined to its +/// chapter + manga, for the "queued chapters" admin view. +#[derive(Debug, Clone, Serialize, FromRow)] +pub struct ActiveJob { + pub id: Uuid, + pub chapter_id: Option, + pub manga_id: Option, + pub manga_title: Option, + pub chapter_number: Option, + /// `"pending"` or `"running"`. + pub state: String, + pub attempts: i32, + pub max_attempts: i32, + pub updated_at: DateTime, +} + +/// Paginated list of `pending`/`running` chapter-content jobs (which +/// chapters of which mangas are queued or being crawled). Running first, +/// then by scheduled order. `search` filters on manga title. +pub async fn list_active_jobs( + pool: &PgPool, + search: Option<&str>, + limit: i64, + offset: i64, +) -> sqlx::Result<(Vec, i64)> { + let search_pat = search + .map(|s| format!("%{}%", s.trim())) + .filter(|p| p.len() > 2); + + let items: Vec = sqlx::query_as( + r#" + SELECT + cj.id, + (cj.payload->>'chapter_id')::uuid AS chapter_id, + c.manga_id AS manga_id, + m.title AS manga_title, + c.number AS chapter_number, + cj.state, + cj.attempts, + cj.max_attempts, + cj.updated_at + FROM crawler_jobs cj + LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid + LEFT JOIN mangas m ON m.id = c.manga_id + WHERE cj.state IN ('pending','running') + AND cj.payload->>'kind' = 'sync_chapter_content' + AND ($1::text IS NULL OR m.title ILIKE $1) + ORDER BY (cj.state = 'running') DESC, cj.scheduled_at, cj.created_at + LIMIT $2 OFFSET $3 + "#, + ) + .bind(&search_pat) + .bind(limit) + .bind(offset) + .fetch_all(pool) + .await?; + + let total: i64 = sqlx::query_scalar( + r#" + SELECT COUNT(*) + FROM crawler_jobs cj + LEFT JOIN chapters c ON c.id = (cj.payload->>'chapter_id')::uuid + LEFT JOIN mangas m ON m.id = c.manga_id + WHERE cj.state IN ('pending','running') + AND cj.payload->>'kind' = 'sync_chapter_content' + AND ($1::text IS NULL OR m.title ILIKE $1) + "#, + ) + .bind(&search_pat) + .fetch_one(pool) + .await?; + + Ok((items, total)) +} + +/// A manga whose cover is still missing (queued for cover fetch). +#[derive(Debug, Clone, Serialize, FromRow)] +pub struct MissingCoverRow { + pub manga_id: Uuid, + pub manga_title: String, +} + +/// Count mangas with no cover yet but a live source row — the cover +/// backlog the metadata pass + backfill drain. +pub async fn count_missing_covers(pool: &PgPool) -> sqlx::Result { + sqlx::query_scalar( + r#" + SELECT COUNT(*) FROM mangas m + WHERE m.cover_image_path IS NULL + AND EXISTS ( + SELECT 1 FROM manga_sources ms + WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL + ) + "#, + ) + .fetch_one(pool) + .await +} + +/// Paginated list of mangas queued for a cover fetch (no cover yet + a live +/// source), with titles. `search` filters on title. Freshest source first. +pub async fn list_missing_cover_mangas( + pool: &PgPool, + search: Option<&str>, + limit: i64, + offset: i64, +) -> sqlx::Result<(Vec, i64)> { + let search_pat = search + .map(|s| format!("%{}%", s.trim())) + .filter(|p| p.len() > 2); + + let items: Vec = sqlx::query_as( + r#" + SELECT m.id AS manga_id, m.title AS manga_title + FROM mangas m + WHERE m.cover_image_path IS NULL + AND EXISTS ( + SELECT 1 FROM manga_sources ms + WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL + ) + AND ($1::text IS NULL OR m.title ILIKE $1) + ORDER BY m.updated_at DESC + LIMIT $2 OFFSET $3 + "#, + ) + .bind(&search_pat) + .bind(limit) + .bind(offset) + .fetch_all(pool) + .await?; + + let total: i64 = sqlx::query_scalar( + r#" + SELECT COUNT(*) FROM mangas m + WHERE m.cover_image_path IS NULL + AND EXISTS ( + SELECT 1 FROM manga_sources ms + WHERE ms.manga_id = m.id AND ms.dropped_at IS NULL + ) + AND ($1::text IS NULL OR m.title ILIKE $1) + "#, + ) + .bind(&search_pat) + .fetch_one(pool) + .await?; + + Ok((items, total)) +} + +/// Scope of a dead-job requeue. +#[derive(Debug, Clone)] +pub enum RequeueScope { + /// Every dead job. + All, + /// Dead jobs whose chapter belongs to this manga. + Manga(Uuid), + /// Dead jobs for a single chapter. + Chapter(Uuid), + /// A single dead job by its id. + Job(Uuid), +} + +/// Requeue dead jobs back to `pending` with a fresh attempt budget. This is +/// an explicit operator override, so it bypasses the dead-letter quarantine +/// the enqueue helpers honour (we act directly on the row). Returns the +/// number of rows requeued. +/// +/// Two invariants protect the partial unique dedup index +/// `crawler_jobs_chapter_content_dedup_idx` (one `pending|running` +/// sync_chapter_content job per chapter): +/// 1. A chapter that already has a live (`pending|running`) job is +/// skipped entirely (`NO_LIVE_DUP`). +/// 2. When a chapter has *multiple* dead jobs, only the newest is +/// revived (`DISTINCT ON` the chapter key) — without this, flipping +/// two dead rows for the same chapter to `pending` in one statement +/// would violate the index and abort the whole requeue. Non-chapter +/// jobs fall back to their row id so each stays distinct. +pub async fn requeue_dead_jobs(pool: &PgPool, scope: RequeueScope) -> sqlx::Result { + // One full-shape SQL string per scope. Previously the scope + // predicate was spliced via `format!()` from a `&'static str` + // match — structurally safe today but fragile against a later + // refactor accidentally interpolating a non-literal. Four fixed + // queries cost a few duplicated lines but are immune to that + // class of bug, and each can be reviewed independently. The CTE + // body (DISTINCT ON dedup + NOT EXISTS guard) is identical + // everywhere so the duplication is mechanical, not semantic. + let q = match scope { + RequeueScope::All => sqlx::query(REQUEUE_DEAD_SQL_ALL), + RequeueScope::Manga(id) => sqlx::query(REQUEUE_DEAD_SQL_MANGA).bind(id), + RequeueScope::Chapter(id) => sqlx::query(REQUEUE_DEAD_SQL_CHAPTER).bind(id), + RequeueScope::Job(id) => sqlx::query(REQUEUE_DEAD_SQL_JOB).bind(id), + }; + Ok(q.execute(pool).await?.rows_affected()) +} + +/// Common shell of the requeue CTE. Each scope variant inlines the +/// shell and substitutes its own WHERE clause; the duplication is the +/// price of avoiding runtime SQL string assembly. The dedup guarantees +/// documented on [`requeue_dead_jobs`] live in the `DISTINCT ON` and +/// the `NOT EXISTS` block — keep those identical across the four +/// constants below when editing. +const REQUEUE_DEAD_SQL_ALL: &str = r#" + WITH pick AS ( + SELECT DISTINCT ON (COALESCE(cj.payload->>'chapter_id', cj.id::text)) cj.id + FROM crawler_jobs cj + WHERE cj.state = 'dead' + AND NOT EXISTS ( + SELECT 1 FROM crawler_jobs live + WHERE live.payload->>'kind' = 'sync_chapter_content' + AND live.payload->>'chapter_id' = cj.payload->>'chapter_id' + AND live.state IN ('pending','running') + ) + ORDER BY COALESCE(cj.payload->>'chapter_id', cj.id::text), cj.updated_at DESC + ) + UPDATE crawler_jobs + SET state = 'pending', attempts = 0, leased_until = NULL, + last_error = NULL, scheduled_at = now(), updated_at = now() + FROM pick + WHERE crawler_jobs.id = pick.id +"#; + +const REQUEUE_DEAD_SQL_MANGA: &str = r#" + WITH pick AS ( + SELECT DISTINCT ON (COALESCE(cj.payload->>'chapter_id', cj.id::text)) cj.id + FROM crawler_jobs cj + WHERE cj.state = 'dead' + AND (cj.payload->>'chapter_id')::uuid IN + (SELECT id FROM chapters WHERE manga_id = $1) + AND NOT EXISTS ( + SELECT 1 FROM crawler_jobs live + WHERE live.payload->>'kind' = 'sync_chapter_content' + AND live.payload->>'chapter_id' = cj.payload->>'chapter_id' + AND live.state IN ('pending','running') + ) + ORDER BY COALESCE(cj.payload->>'chapter_id', cj.id::text), cj.updated_at DESC + ) + UPDATE crawler_jobs + SET state = 'pending', attempts = 0, leased_until = NULL, + last_error = NULL, scheduled_at = now(), updated_at = now() + FROM pick + WHERE crawler_jobs.id = pick.id +"#; + +const REQUEUE_DEAD_SQL_CHAPTER: &str = r#" + WITH pick AS ( + SELECT DISTINCT ON (COALESCE(cj.payload->>'chapter_id', cj.id::text)) cj.id + FROM crawler_jobs cj + WHERE cj.state = 'dead' + AND (cj.payload->>'chapter_id')::uuid = $1 + AND NOT EXISTS ( + SELECT 1 FROM crawler_jobs live + WHERE live.payload->>'kind' = 'sync_chapter_content' + AND live.payload->>'chapter_id' = cj.payload->>'chapter_id' + AND live.state IN ('pending','running') + ) + ORDER BY COALESCE(cj.payload->>'chapter_id', cj.id::text), cj.updated_at DESC + ) + UPDATE crawler_jobs + SET state = 'pending', attempts = 0, leased_until = NULL, + last_error = NULL, scheduled_at = now(), updated_at = now() + FROM pick + WHERE crawler_jobs.id = pick.id +"#; + +const REQUEUE_DEAD_SQL_JOB: &str = r#" + WITH pick AS ( + SELECT DISTINCT ON (COALESCE(cj.payload->>'chapter_id', cj.id::text)) cj.id + FROM crawler_jobs cj + WHERE cj.state = 'dead' + AND cj.id = $1 + AND NOT EXISTS ( + SELECT 1 FROM crawler_jobs live + WHERE live.payload->>'kind' = 'sync_chapter_content' + AND live.payload->>'chapter_id' = cj.payload->>'chapter_id' + AND live.state IN ('pending','running') + ) + ORDER BY COALESCE(cj.payload->>'chapter_id', cj.id::text), cj.updated_at DESC + ) + UPDATE crawler_jobs + SET state = 'pending', attempts = 0, leased_until = NULL, + last_error = NULL, scheduled_at = now(), updated_at = now() + FROM pick + WHERE crawler_jobs.id = pick.id +"#; + +/// `crawler_state` key under which the runtime session value (an admin +/// pushed PHPSESSID) is persisted. Survives a backend restart so a +/// mid-day refresh isn't lost. +const STATE_KEY_RUNTIME_SESSION: &str = "runtime_session"; + +/// Read the persisted runtime PHPSESSID (if any). The payload shape is +/// `{ "phpsessid": "" }`; anything else returns `None`. +pub async fn runtime_session_load(pool: &PgPool) -> sqlx::Result> { + let row: Option = + sqlx::query_scalar("SELECT value FROM crawler_state WHERE key = $1") + .bind(STATE_KEY_RUNTIME_SESSION) + .fetch_optional(pool) + .await?; + Ok(row.and_then(|v| { + v.get("phpsessid") + .and_then(|s| s.as_str()) + .map(|s| s.to_string()) + })) +} + +/// Persist a fresh runtime PHPSESSID, replacing any previous value. +pub async fn runtime_session_persist(pool: &PgPool, sid: &str) -> sqlx::Result<()> { + sqlx::query( + "INSERT INTO crawler_state (key, value, updated_at) \ + VALUES ($1, $2, now()) \ + ON CONFLICT (key) DO UPDATE \ + SET value = EXCLUDED.value, updated_at = now()", + ) + .bind(STATE_KEY_RUNTIME_SESSION) + .bind(serde_json::json!({ "phpsessid": sid })) + .execute(pool) + .await?; + Ok(()) +} + +/// Count crawler jobs grouped by state — drives the dashboard queue +/// gauges. Returns `(pending, running, dead)`. +pub async fn job_state_counts(pool: &PgPool) -> sqlx::Result<(i64, i64, i64)> { + let rows: Vec<(String, i64)> = + sqlx::query_as("SELECT state, COUNT(*) FROM crawler_jobs GROUP BY state") + .fetch_all(pool) + .await?; + let mut pending = 0; + let mut running = 0; + let mut dead = 0; + for (state, n) in rows { + match state.as_str() { + "pending" => pending = n, + "running" => running = n, + "dead" => dead = n, + _ => {} + } + } + Ok((pending, running, dead)) +} + diff --git a/backend/src/storage/local.rs b/backend/src/storage/local.rs index fed5af8..cfa9345 100644 --- a/backend/src/storage/local.rs +++ b/backend/src/storage/local.rs @@ -1,10 +1,12 @@ use std::path::{Path, PathBuf}; use async_trait::async_trait; +use futures_util::StreamExt as _; use tokio::fs; +use tokio::io::AsyncWriteExt as _; use tokio_util::io::ReaderStream; -use super::{Storage, StorageError, StreamingFile}; +use super::{PutByteStream, Storage, StorageError, StreamingFile}; pub struct LocalStorage { root: PathBuf, @@ -45,6 +47,49 @@ impl Storage for LocalStorage { Ok(()) } + async fn put_stream( + &self, + key: &str, + mut stream: PutByteStream<'_>, + ) -> Result { + let path = self.resolve(key)?; + if let Some(parent) = path.parent() { + fs::create_dir_all(parent).await?; + } + // Atomic install via temp + rename. A failure mid-stream + // removes the temp so nothing is visible at `key`. The temp + // name uses a UUID suffix so concurrent puts of the same key + // (e.g. two workers racing a retry) don't clobber each + // other's in-progress file before the rename. + let tmp = path.with_extension(format!( + "{}.tmp.{}", + path.extension().and_then(|e| e.to_str()).unwrap_or(""), + uuid::Uuid::new_v4().simple() + )); + let mut written: u64 = 0; + let result: Result<(), StorageError> = async { + let mut f = fs::File::create(&tmp).await?; + while let Some(chunk) = stream.next().await { + let chunk = chunk?; + f.write_all(&chunk).await?; + written = written.saturating_add(chunk.len() as u64); + } + // fsync before rename so a power-loss can't leave a + // zero-byte file at the destination. + f.sync_all().await?; + fs::rename(&tmp, &path).await?; + Ok(()) + } + .await; + if let Err(e) = result { + // Best-effort cleanup; ignore "no such file" if the temp + // was never created. + let _ = fs::remove_file(&tmp).await; + return Err(e); + } + Ok(written) + } + async fn get(&self, key: &str) -> Result, StorageError> { let path = self.resolve(key)?; match fs::read(&path).await { @@ -142,6 +187,52 @@ mod tests { )); } + #[tokio::test] + async fn put_stream_writes_full_body_and_removes_temp_on_error() { + use bytes::Bytes; + use futures_util::stream; + + let dir = tempdir().unwrap(); + let s = LocalStorage::new(dir.path()); + + // Success: a 3-chunk stream writes the concatenation and + // returns the right byte count. + let chunks: Vec> = vec![ + Ok(Bytes::from_static(b"alpha-")), + Ok(Bytes::from_static(b"beta-")), + Ok(Bytes::from_static(b"gamma")), + ]; + let bytes_written = s + .put_stream("streamed/ok.bin", Box::pin(stream::iter(chunks))) + .await + .unwrap(); + assert_eq!(bytes_written, b"alpha-beta-gamma".len() as u64); + assert_eq!(s.get("streamed/ok.bin").await.unwrap(), b"alpha-beta-gamma"); + + // Failure mid-stream: nothing is visible at the destination + // and no .tmp file is left behind. + let chunks: Vec> = vec![ + Ok(Bytes::from_static(b"good")), + Err(StorageError::Io(std::io::Error::other("boom"))), + ]; + let err = s + .put_stream("streamed/bad.bin", Box::pin(stream::iter(chunks))) + .await + .unwrap_err(); + assert!(matches!(err, StorageError::Io(_))); + assert!(matches!( + s.get("streamed/bad.bin").await, + Err(StorageError::NotFound) + )); + // No stray temp file in the streamed/ directory. + let entries: Vec<_> = std::fs::read_dir(dir.path().join("streamed")) + .unwrap() + .filter_map(|e| e.ok()) + .map(|e| e.file_name().to_string_lossy().to_string()) + .collect(); + assert_eq!(entries, vec!["ok.bin"]); + } + #[tokio::test] async fn get_stream_emits_multiple_chunks_for_large_files() { use futures_util::StreamExt as _; diff --git a/backend/src/storage/mod.rs b/backend/src/storage/mod.rs index c00d803..af6fc06 100644 --- a/backend/src/storage/mod.rs +++ b/backend/src/storage/mod.rs @@ -31,6 +31,13 @@ pub enum StorageError { /// object-safe regardless of the concrete reader behind it. pub type ByteStream = Pin> + Send>>; +/// Boxed byte stream accepted by `Storage::put_stream`. The item type +/// is fallible so a producer (e.g. an HTTP body) can surface a transport +/// error mid-stream without breaking the trait shape; the storage impl +/// is responsible for not installing a partial blob on such an error. +pub type PutByteStream<'a> = + Pin> + Send + 'a>>; + pub struct StreamingFile { pub stream: ByteStream, pub size_bytes: u64, @@ -39,6 +46,34 @@ pub struct StreamingFile { #[async_trait] pub trait Storage: Send + Sync { async fn put(&self, key: &str, bytes: &[u8]) -> Result<(), StorageError>; + /// Stream a blob to storage without holding the entire body in + /// memory. The chapter-content download path uses this so peak + /// memory stays at one chunk per concurrent dispatch (not one full + /// page image). The contract is atomic: a stream that errors mid-way + /// must leave nothing visible at `key` — implementations should + /// write to a temp location and rename only on the successful + /// drain. Returns the total bytes written on success. + /// + /// The default implementation buffers the stream into memory and + /// calls `put`, so backends without a native streaming write still + /// satisfy the contract (at the cost of peak memory). LocalStorage + /// overrides this to do a temp-file rename; a future S3Storage + /// would override with a multi-part upload. + async fn put_stream( + &self, + key: &str, + mut stream: PutByteStream<'_>, + ) -> Result { + use futures_util::StreamExt as _; + let mut buf: Vec = Vec::new(); + while let Some(chunk) = stream.next().await { + let chunk = chunk?; + buf.extend_from_slice(&chunk); + } + let len = buf.len() as u64; + self.put(key, &buf).await?; + Ok(len) + } /// Reads the entire blob into memory. Convenient for small assets /// (covers, thumbnails). For pages and other large blobs, use /// `get_stream` so axum can pipe bytes straight to the client. diff --git a/backend/tests/api_admin_crawler.rs b/backend/tests/api_admin_crawler.rs new file mode 100644 index 0000000..1fdb3be --- /dev/null +++ b/backend/tests/api_admin_crawler.rs @@ -0,0 +1,654 @@ +//! Integration tests for the admin crawler observability/control API. +//! +//! The default test harness wires `AppState.crawler = None` (no daemon), +//! so the *control* endpoints return 503 and the *read* endpoints that +//! work off the DB (status shell, dead-jobs list/requeue) still function. +//! This is exactly the production "daemon disabled" posture. + +mod common; + +use std::time::Duration; + +use axum::http::StatusCode; +use axum::Router; +use http_body_util::BodyExt; +use serde_json::json; +use sqlx::PgPool; +use tower::ServiceExt; +use uuid::Uuid; + +use common::{ + body_json, get, get_with_cookie, harness, harness_with_admin_origins, + post_json_with_cookie, post_json_with_cookie_origin, register_user, +}; + +async fn seed_admin(pool: &PgPool, app: &Router) -> String { + let (username, cookie) = register_user(app).await; + let u = mangalord::repo::user::find_by_username(pool, &username) + .await + .unwrap() + .unwrap(); + mangalord::repo::user::set_is_admin_unchecked(pool, u.id, true) + .await + .unwrap(); + cookie +} + +async fn seed_dead_job(pool: &PgPool, title: &str) -> Uuid { + let manga_id = Uuid::new_v4(); + let chapter_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, $2)") + .bind(manga_id) + .bind(title) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO chapters (id, manga_id, number) VALUES ($1, $2, 1)") + .bind(chapter_id) + .bind(manga_id) + .execute(pool) + .await + .unwrap(); + let job_id = Uuid::new_v4(); + sqlx::query( + "INSERT INTO crawler_jobs (id, payload, state, attempts, last_error) \ + VALUES ($1, $2, 'dead', 5, 'boom')", + ) + .bind(job_id) + .bind(json!({ + "kind": "sync_chapter_content", + "source_id": "target", + "chapter_id": chapter_id, + "source_chapter_key": "k", + })) + .execute(pool) + .await + .unwrap(); + job_id +} + +/// Seed a chapter-content job in a given state ('pending'/'running'). +async fn seed_job(pool: &PgPool, title: &str, state: &str) { + let manga_id = Uuid::new_v4(); + let chapter_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, $2)") + .bind(manga_id) + .bind(title) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO chapters (id, manga_id, number) VALUES ($1, $2, 1)") + .bind(chapter_id) + .bind(manga_id) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO crawler_jobs (id, payload, state) VALUES ($1, $2, $3)") + .bind(Uuid::new_v4()) + .bind(json!({ + "kind": "sync_chapter_content", + "source_id": "target", + "chapter_id": chapter_id, + "source_chapter_key": "k", + })) + .bind(state) + .execute(pool) + .await + .unwrap(); +} + +/// Seed a manga with no cover + a live source row (queued for cover fetch). +async fn seed_missing_cover(pool: &PgPool, title: &str) { + let manga_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title, cover_image_path) VALUES ($1, $2, NULL)") + .bind(manga_id) + .bind(title) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO sources (id, name, base_url) VALUES ('target','T','http://x') ON CONFLICT DO NOTHING") + .execute(pool) + .await + .unwrap(); + sqlx::query( + "INSERT INTO manga_sources (source_id, source_manga_key, manga_id, source_url) \ + VALUES ('target', $1, $2, 'http://x/m')", + ) + .bind(format!("k-{manga_id}")) + .bind(manga_id) + .execute(pool) + .await + .unwrap(); +} + +#[sqlx::test(migrations = "./migrations")] +async fn active_jobs_and_covers_lists_over_http(pool: PgPool) { + seed_job(&pool, "Naruto", "pending").await; + seed_job(&pool, "Bleach", "running").await; + seed_missing_cover(&pool, "One Piece").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + + // Queued/active chapters. + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/active-jobs", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_json(resp).await; + assert_eq!(body["page"]["total"], 2); + + // Queued covers. + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/covers", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_json(resp).await; + assert_eq!(body["page"]["total"], 1); + assert_eq!(body["items"][0]["manga_title"], "One Piece"); + + // Both are admin-gated. + let (_u, plain) = register_user(&h.app).await; + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/active-jobs", &plain)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::FORBIDDEN); +} + +#[sqlx::test(migrations = "./migrations")] +async fn get_status_requires_admin(pool: PgPool) { + let h = harness(pool); + // Unauthenticated → 401. + let resp = h.app.clone().oneshot(get("/api/v1/admin/crawler")).await.unwrap(); + assert_eq!(resp.status(), StatusCode::UNAUTHORIZED); + + // Authenticated non-admin → 403. + let (_u, cookie) = register_user(&h.app).await; + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::FORBIDDEN); +} + +#[sqlx::test(migrations = "./migrations")] +async fn get_status_reports_disabled_daemon_with_queue_counts(pool: PgPool) { + seed_dead_job(&pool, "Naruto").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_json(resp).await; + assert_eq!(body["daemon"], "disabled"); + assert_eq!(body["queue"]["dead"], 1); + assert_eq!(body["browser"], "down"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn control_endpoints_return_503_when_daemon_disabled(pool: PgPool) { + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + for uri in [ + "/api/v1/admin/crawler/run", + "/api/v1/admin/crawler/browser/restart", + "/api/v1/admin/crawler/session/clear-expired", + ] { + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie(uri, json!({}), &cookie)) + .await + .unwrap(); + assert_eq!( + resp.status(), + StatusCode::SERVICE_UNAVAILABLE, + "{uri} should be 503 when daemon disabled" + ); + } +} + +#[sqlx::test(migrations = "./migrations")] +async fn status_stream_requires_admin(pool: PgPool) { + let h = harness(pool); + // Unauthenticated → 401. + let resp = h + .app + .clone() + .oneshot(get("/api/v1/admin/crawler/stream")) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::UNAUTHORIZED); + // Non-admin → 403. + let (_u, cookie) = register_user(&h.app).await; + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/stream", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::FORBIDDEN); +} + +#[sqlx::test(migrations = "./migrations")] +async fn status_stream_emits_initial_event(pool: PgPool) { + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/stream", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let ct = resp + .headers() + .get(axum::http::header::CONTENT_TYPE) + .and_then(|v| v.to_str().ok()) + .unwrap_or_default() + .to_string(); + assert!(ct.starts_with("text/event-stream"), "content-type was {ct:?}"); + + // Accumulate frames (the immediate snapshot may arrive split across + // frames) until the status payload appears, with an overall timeout so + // the never-ending stream can't hang the test. + let mut body = resp.into_body(); + let mut acc = String::new(); + let deadline = tokio::time::timeout(Duration::from_secs(5), async { + loop { + let Some(frame) = body.frame().await else { break }; + if let Ok(data) = frame.expect("frame ok").into_data() { + acc.push_str(&String::from_utf8_lossy(&data)); + if acc.contains("\"daemon\"") { + break; + } + } + } + }) + .await; + assert!(deadline.is_ok(), "did not receive status within 5s; got: {acc:?}"); + assert!(acc.contains("\"daemon\""), "missing status payload: {acc}"); + assert!(acc.contains("status"), "missing SSE event name: {acc}"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn mutating_endpoints_reject_non_admin(pool: PgPool) { + let h = harness(pool); + // A logged-in non-admin must be forbidden from a mutating endpoint. + let (_u, cookie) = register_user(&h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "all" }), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::FORBIDDEN); +} + +// --------------------------------------------------------------------------- +// CSRF Origin/Referer allowlist (T2) +// --------------------------------------------------------------------------- + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_rejects_mutation_with_cross_origin_header(pool: PgPool) { + let h = harness_with_admin_origins( + pool.clone(), + vec!["http://localhost:3000".to_string()], + ); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie_origin( + "/api/v1/admin/crawler/session/clear-expired", + json!({}), + &cookie, + Some("https://evil.example.com"), + None, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::FORBIDDEN); +} + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_allows_mutation_with_allowed_origin(pool: PgPool) { + let h = harness_with_admin_origins( + pool.clone(), + vec!["http://localhost:3000".to_string()], + ); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie_origin( + "/api/v1/admin/crawler/session/clear-expired", + json!({}), + &cookie, + Some("http://localhost:3000"), + None, + )) + .await + .unwrap(); + // Daemon disabled in the harness → 503 from the handler; the CSRF + // gate must have let the request through before that response. + assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE); +} + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_allows_mutation_without_origin_or_referer(pool: PgPool) { + // curl/server-to-server callers send neither — they can't be a CSRF + // vector since there's no third-party browser context. + let h = harness_with_admin_origins( + pool.clone(), + vec!["http://localhost:3000".to_string()], + ); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/session/clear-expired", + json!({}), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE); +} + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_falls_back_to_referer_when_origin_missing(pool: PgPool) { + let h = harness_with_admin_origins( + pool.clone(), + vec!["http://localhost:3000".to_string()], + ); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie_origin( + "/api/v1/admin/crawler/session/clear-expired", + json!({}), + &cookie, + None, + Some("http://localhost:3000/admin/crawler"), + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE); +} + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_skipped_on_safe_methods(pool: PgPool) { + let h = harness_with_admin_origins( + pool.clone(), + vec!["http://localhost:3000".to_string()], + ); + let cookie = seed_admin(&pool, &h.app).await; + // GET from a hostile origin is fine — browsers can't mutate via GET. + let resp = h + .app + .clone() + .oneshot(common::get_with_cookie_origin( + "/api/v1/admin/crawler", + &cookie, + Some("https://evil.example.com"), + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); +} + +#[sqlx::test(migrations = "./migrations")] +async fn csrf_disabled_when_allowlist_empty(pool: PgPool) { + // Default harness has admin_allowed_origins = empty → operator + // opt-out documented in .env.example. A cross-origin POST passes + // through to the handler. + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie_origin( + "/api/v1/admin/crawler/session/clear-expired", + json!({}), + &cookie, + Some("https://evil.example.com"), + None, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE); +} + +// --------------------------------------------------------------------------- +// Cache-Control: no-store on admin responses (S3) +// --------------------------------------------------------------------------- + +#[sqlx::test(migrations = "./migrations")] +async fn admin_responses_have_no_store(pool: PgPool) { + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let cc = resp + .headers() + .get(axum::http::header::CACHE_CONTROL) + .and_then(|v| v.to_str().ok()) + .unwrap_or_default() + .to_string(); + assert!( + cc.contains("no-store"), + "admin response missing Cache-Control: no-store (got {cc:?})" + ); +} + +#[sqlx::test(migrations = "./migrations")] +async fn non_admin_responses_unaffected_by_no_store(pool: PgPool) { + let h = harness(pool); + let resp = h + .app + .clone() + .oneshot(get("/api/v1/health")) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let cc = resp + .headers() + .get(axum::http::header::CACHE_CONTROL) + .and_then(|v| v.to_str().ok()); + assert!( + cc.is_none() || !cc.unwrap_or_default().contains("no-store"), + "non-admin response should not get no-store (got {cc:?})" + ); +} + +// --------------------------------------------------------------------------- +// scope=all confirm:true guard (S1) +// --------------------------------------------------------------------------- + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_all_without_confirm_rejected(pool: PgPool) { + seed_dead_job(&pool, "X").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "all" }), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::BAD_REQUEST); + // The dead row must NOT have been touched. + let state: String = sqlx::query_scalar("SELECT state FROM crawler_jobs LIMIT 1") + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(state, "dead"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_all_with_confirm_flips_dead_pile(pool: PgPool) { + seed_dead_job(&pool, "X").await; + seed_dead_job(&pool, "Y").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "all", "confirm": true }), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let pending: i64 = sqlx::query_scalar( + "SELECT COUNT(*) FROM crawler_jobs WHERE state = 'pending'", + ) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(pending, 2); +} + +// --------------------------------------------------------------------------- +// admin_audit target_id + PHPSESSID fingerprint (S2) +// --------------------------------------------------------------------------- + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_chapter_audit_records_chapter_target_id(pool: PgPool) { + let job_id = seed_dead_job(&pool, "Bleach").await; + let chapter_id: Uuid = sqlx::query_scalar( + "SELECT (payload->>'chapter_id')::uuid FROM crawler_jobs WHERE id = $1", + ) + .bind(job_id) + .fetch_one(&pool) + .await + .unwrap(); + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "chapter", "chapter_id": chapter_id }), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let (target_kind, target_id): (String, Option) = sqlx::query_as( + "SELECT target_kind, target_id FROM admin_audit \ + WHERE action = 'crawler_dead_jobs_requeue' \ + ORDER BY created_at DESC LIMIT 1", + ) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(target_kind, "chapter"); + assert_eq!(target_id, Some(chapter_id)); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_all_audit_omits_target_id_but_logs_count(pool: PgPool) { + seed_dead_job(&pool, "X").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + let _ = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "all", "confirm": true }), + &cookie, + )) + .await + .unwrap(); + let (target_kind, target_id, payload): (String, Option, serde_json::Value) = + sqlx::query_as( + "SELECT target_kind, target_id, payload FROM admin_audit \ + WHERE action = 'crawler_dead_jobs_requeue' \ + ORDER BY created_at DESC LIMIT 1", + ) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(target_kind, "crawler"); + assert_eq!(target_id, None); + assert_eq!(payload["scope"], "all"); + assert_eq!(payload["requeued"], 1); +} + +#[sqlx::test(migrations = "./migrations")] +async fn dead_jobs_list_and_requeue_over_http(pool: PgPool) { + let job_id = seed_dead_job(&pool, "Bleach").await; + let h = harness(pool.clone()); + let cookie = seed_admin(&pool, &h.app).await; + + // List. + let resp = h + .app + .clone() + .oneshot(get_with_cookie("/api/v1/admin/crawler/dead-jobs", &cookie)) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_json(resp).await; + assert_eq!(body["page"]["total"], 1); + assert_eq!(body["items"][0]["manga_title"], "Bleach"); + + // Requeue the single job. + let resp = h + .app + .clone() + .oneshot(post_json_with_cookie( + "/api/v1/admin/crawler/dead-jobs/requeue", + json!({ "scope": "job", "job_id": job_id }), + &cookie, + )) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_json(resp).await; + assert_eq!(body["requeued"], 1); + + let state: String = sqlx::query_scalar("SELECT state FROM crawler_jobs WHERE id = $1") + .bind(job_id) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(state, "pending"); +} diff --git a/backend/tests/api_admin_role.rs b/backend/tests/api_admin_role.rs index a88f503..05d51ed 100644 --- a/backend/tests/api_admin_role.rs +++ b/backend/tests/api_admin_role.rs @@ -50,6 +50,8 @@ fn admin_test_router(pool: PgPool) -> (Router, TempDir) { upload: UploadConfig::default(), auth_limiter, resync: None, + crawler: None, + admin_allowed_origins: Arc::new(Vec::new()), }; let app = Router::new() .nest("/api/v1", api::routes()) diff --git a/backend/tests/common/mod.rs b/backend/tests/common/mod.rs index 12315b3..78e75e3 100644 --- a/backend/tests/common/mod.rs +++ b/backend/tests/common/mod.rs @@ -17,7 +17,7 @@ use tower::ServiceExt; use mangalord::app::{router, AppState}; use mangalord::auth::rate_limit::AuthRateLimiter; use mangalord::config::{AuthConfig, UploadConfig}; -use mangalord::storage::{LocalStorage, Storage, StorageError, StreamingFile}; +use mangalord::storage::{LocalStorage, PutByteStream, Storage, StorageError, StreamingFile}; use async_trait::async_trait; use std::sync::atomic::{AtomicUsize, Ordering}; @@ -78,6 +78,10 @@ fn harness_with_auth_config( // handlers return 503 in this config. Tests that need a stub // resync service swap it in via `harness_with_resync`. resync: None, + crawler: None, + // Empty allowlist = CSRF check skipped. The CSRF-specific test + // harness `harness_with_admin_origins` overrides this. + admin_allowed_origins: Arc::new(Vec::new()), }; Harness { app: router(state), _storage_dir: storage_dir } } @@ -152,6 +156,37 @@ pub fn harness_with_resync( }, auth_limiter, resync: Some(resync), + crawler: None, + admin_allowed_origins: Arc::new(Vec::new()), + }; + Harness { + app: router(state), + _storage_dir: storage_dir, + } +} + +/// Like [`harness`] but configures an admin CSRF allowlist so the +/// `/admin/*` mutating endpoints reject cross-origin browser POSTs. +/// Used by the admin CSRF integration tests. +pub fn harness_with_admin_origins(pool: PgPool, origins: Vec) -> Harness { + let storage_dir = tempfile::tempdir().expect("tempdir"); + let storage = Arc::new(LocalStorage::new(storage_dir.path())); + let auth_limiter = Arc::new(AuthRateLimiter::new(Default::default())); + let state = AppState { + db: pool, + storage, + auth: AuthConfig { + cookie_secure: false, + ..AuthConfig::default() + }, + upload: UploadConfig { + max_request_bytes: 4 * 1024 * 1024, + max_file_bytes: 256 * 1024, + }, + auth_limiter, + resync: None, + crawler: None, + admin_allowed_origins: Arc::new(origins), }; Harness { app: router(state), @@ -189,6 +224,22 @@ impl Storage for FailingStorage { } self.inner.put(key, bytes).await } + async fn put_stream( + &self, + key: &str, + stream: PutByteStream<'_>, + ) -> Result { + // Count put_stream towards the same fail-index so tests that + // expect "the Nth put fails" don't care which entry point + // the caller took. + let n = self.counter.fetch_add(1, Ordering::SeqCst); + if n == self.fail_on_put_index { + return Err(StorageError::Io(std::io::Error::other( + "FailingStorage: injected put_stream failure", + ))); + } + self.inner.put_stream(key, stream).await + } async fn get(&self, key: &str) -> Result, StorageError> { self.inner.get(key).await } @@ -251,6 +302,44 @@ pub fn post_json_with_cookie( .unwrap() } +/// Same as [`post_json_with_cookie`] but also attaches `Origin` (and +/// optionally `Referer`) headers. Used by the admin CSRF tests to drive +/// the cross-origin reject + allowed-origin accept paths. +pub fn post_json_with_cookie_origin( + uri: &str, + body: serde_json::Value, + cookie: &str, + origin: Option<&str>, + referer: Option<&str>, +) -> Request { + let mut b = Request::builder() + .method("POST") + .uri(uri) + .header(header::CONTENT_TYPE, "application/json") + .header(header::COOKIE, cookie); + if let Some(o) = origin { + b = b.header(header::ORIGIN, o); + } + if let Some(r) = referer { + b = b.header(header::REFERER, r); + } + b.body(Body::from(body.to_string())).unwrap() +} + +pub fn get_with_cookie_origin( + uri: &str, + cookie: &str, + origin: Option<&str>, +) -> Request { + let mut b = Request::builder() + .uri(uri) + .header(header::COOKIE, cookie); + if let Some(o) = origin { + b = b.header(header::ORIGIN, o); + } + b.body(Body::empty()).unwrap() +} + pub fn post_json_with_bearer( uri: &str, body: serde_json::Value, diff --git a/backend/tests/crawler_daemon.rs b/backend/tests/crawler_daemon.rs index 16b308d..f2d7a2e 100644 --- a/backend/tests/crawler_daemon.rs +++ b/backend/tests/crawler_daemon.rs @@ -40,6 +40,8 @@ fn make_cfg( tz: Tz::UTC, retention_days: 7, session_expired, + status: mangalord::crawler::status::StatusHandle::new(workers), + job_timeout: Duration::from_secs(60), extra_tasks: Vec::new(), } } @@ -88,6 +90,52 @@ impl ChapterDispatcher for PanickingDispatcher { } } +/// Never completes — used to verify the worker's outer dispatch timeout. +struct HangingDispatcher { + seen: AtomicUsize, +} +#[async_trait::async_trait] +impl ChapterDispatcher for HangingDispatcher { + async fn dispatch(&self, _payload: JobPayload) -> anyhow::Result { + self.seen.fetch_add(1, Ordering::AcqRel); + std::future::pending::<()>().await; + unreachable!("hanging dispatcher never resolves"); + } +} + +#[sqlx::test(migrations = "./migrations")] +async fn worker_times_out_a_hung_dispatch_and_acks_failed(pool: PgPool) { + enqueue_chapter_job(&pool).await; + let dispatcher = Arc::new(HangingDispatcher { + seen: AtomicUsize::new(0), + }); + let session_expired = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let cancel = CancellationToken::new(); + let mut cfg = make_cfg(None, dispatcher.clone(), session_expired, 1); + cfg.job_timeout = Duration::from_millis(300); + let handle = daemon::spawn(pool.clone(), cancel.clone(), cfg); + + // The hung job should time out and return to pending with backoff + // (attempts=1 < max=5). Poll for the recorded error. + let mut timed_out = false; + for _ in 0..40 { + let n: i64 = sqlx::query_scalar( + "SELECT COUNT(*) FROM crawler_jobs WHERE last_error = 'dispatch timed out'", + ) + .fetch_one(&pool) + .await + .unwrap(); + if n == 1 { + timed_out = true; + break; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + handle.shutdown().await; + assert!(timed_out, "hung dispatch must be acked failed with a timeout error"); + assert!(dispatcher.seen.load(Ordering::Acquire) >= 1); +} + #[sqlx::test(migrations = "./migrations")] async fn workers_drain_jobs_through_dispatcher(pool: PgPool) { enqueue_chapter_job(&pool).await; diff --git a/backend/tests/crawler_dead_jobs.rs b/backend/tests/crawler_dead_jobs.rs new file mode 100644 index 0000000..5972f4f --- /dev/null +++ b/backend/tests/crawler_dead_jobs.rs @@ -0,0 +1,304 @@ +//! Integration tests for the dead-letter admin queries in +//! `repo::crawler`: listing dead jobs with manga/chapter context and the +//! scoped requeue (all / per-manga / single) used by the admin dashboard. + +use mangalord::repo::crawler::{self, RequeueScope}; +use serde_json::json; +use sqlx::PgPool; +use uuid::Uuid; + +/// Seed a manga with no cover + a live source row (so it's "queued for a +/// cover fetch"). Returns the manga id. +async fn seed_missing_cover(pool: &PgPool, title: &str) -> Uuid { + let manga_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title, cover_image_path) VALUES ($1, $2, NULL)") + .bind(manga_id) + .bind(title) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO sources (id, name, base_url) VALUES ('target', 'T', 'http://x') ON CONFLICT DO NOTHING") + .execute(pool) + .await + .unwrap(); + sqlx::query( + "INSERT INTO manga_sources (source_id, source_manga_key, manga_id, source_url) \ + VALUES ('target', $1, $2, 'http://x/m')", + ) + .bind(format!("k-{manga_id}")) + .bind(manga_id) + .execute(pool) + .await + .unwrap(); + manga_id +} + +/// Seed a manga + chapter and return their ids. +async fn seed_chapter(pool: &PgPool, title: &str, number: i32) -> (Uuid, Uuid) { + let manga_id = Uuid::new_v4(); + let chapter_id = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title) VALUES ($1, $2)") + .bind(manga_id) + .bind(title) + .execute(pool) + .await + .unwrap(); + sqlx::query("INSERT INTO chapters (id, manga_id, number) VALUES ($1, $2, $3)") + .bind(chapter_id) + .bind(manga_id) + .bind(number) + .execute(pool) + .await + .unwrap(); + (manga_id, chapter_id) +} + +/// Insert a crawler_jobs row in a given state for a chapter-content job. +async fn insert_job(pool: &PgPool, chapter_id: Uuid, state: &str, attempts: i32) -> Uuid { + let id = Uuid::new_v4(); + let payload = json!({ + "kind": "sync_chapter_content", + "source_id": "target", + "chapter_id": chapter_id, + "source_chapter_key": "k", + }); + sqlx::query( + "INSERT INTO crawler_jobs (id, payload, state, attempts, last_error) \ + VALUES ($1, $2, $3, $4, 'boom')", + ) + .bind(id) + .bind(payload) + .bind(state) + .bind(attempts) + .execute(pool) + .await + .unwrap(); + id +} + +async fn state_of(pool: &PgPool, id: Uuid) -> String { + sqlx::query_scalar::<_, String>("SELECT state FROM crawler_jobs WHERE id = $1") + .bind(id) + .fetch_one(pool) + .await + .unwrap() +} + +#[sqlx::test(migrations = "./migrations")] +async fn list_dead_jobs_returns_context_and_total(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "Naruto", 700).await; + insert_job(&pool, c1, "dead", 5).await; + // A non-dead job must not appear. + let (_m2, c2) = seed_chapter(&pool, "Bleach", 1).await; + insert_job(&pool, c2, "pending", 0).await; + + let (items, total) = crawler::list_dead_jobs(&pool, None, 50, 0).await.unwrap(); + assert_eq!(total, 1); + assert_eq!(items.len(), 1); + let row = &items[0]; + assert_eq!(row.manga_title.as_deref(), Some("Naruto")); + assert_eq!(row.chapter_number, Some(700)); + assert_eq!(row.attempts, 5); + assert_eq!(row.last_error.as_deref(), Some("boom")); +} + +#[sqlx::test(migrations = "./migrations")] +async fn list_dead_jobs_filters_by_title_search(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "Naruto", 700).await; + insert_job(&pool, c1, "dead", 5).await; + let (_m2, c2) = seed_chapter(&pool, "One Piece", 1).await; + insert_job(&pool, c2, "dead", 5).await; + + let (items, total) = crawler::list_dead_jobs(&pool, Some("piece"), 50, 0) + .await + .unwrap(); + assert_eq!(total, 1); + assert_eq!(items[0].manga_title.as_deref(), Some("One Piece")); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_all_resets_dead_jobs_to_pending(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "A", 1).await; + let (_m2, c2) = seed_chapter(&pool, "B", 1).await; + let j1 = insert_job(&pool, c1, "dead", 5).await; + let j2 = insert_job(&pool, c2, "dead", 5).await; + + let n = crawler::requeue_dead_jobs(&pool, RequeueScope::All) + .await + .unwrap(); + assert_eq!(n, 2); + assert_eq!(state_of(&pool, j1).await, "pending"); + assert_eq!(state_of(&pool, j2).await, "pending"); + let attempts: i32 = sqlx::query_scalar("SELECT attempts FROM crawler_jobs WHERE id = $1") + .bind(j1) + .fetch_one(&pool) + .await + .unwrap(); + assert_eq!(attempts, 0, "attempts reset on requeue"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_by_manga_scopes_to_that_manga(pool: PgPool) { + let (m1, c1) = seed_chapter(&pool, "A", 1).await; + let (_m2, c2) = seed_chapter(&pool, "B", 1).await; + let j1 = insert_job(&pool, c1, "dead", 5).await; + let j2 = insert_job(&pool, c2, "dead", 5).await; + + let n = crawler::requeue_dead_jobs(&pool, RequeueScope::Manga(m1)) + .await + .unwrap(); + assert_eq!(n, 1); + assert_eq!(state_of(&pool, j1).await, "pending"); + assert_eq!(state_of(&pool, j2).await, "dead", "other manga untouched"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_by_chapter_scopes_to_that_chapter(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "A", 1).await; + let (_m2, c2) = seed_chapter(&pool, "A", 2).await; + let j1 = insert_job(&pool, c1, "dead", 5).await; + let j2 = insert_job(&pool, c2, "dead", 5).await; + + let n = crawler::requeue_dead_jobs(&pool, RequeueScope::Chapter(c1)) + .await + .unwrap(); + assert_eq!(n, 1); + assert_eq!(state_of(&pool, j1).await, "pending"); + assert_eq!(state_of(&pool, j2).await, "dead", "other chapter untouched"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_single_job(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "A", 1).await; + let (_m2, c2) = seed_chapter(&pool, "B", 1).await; + let j1 = insert_job(&pool, c1, "dead", 5).await; + let j2 = insert_job(&pool, c2, "dead", 5).await; + + let n = crawler::requeue_dead_jobs(&pool, RequeueScope::Job(j1)) + .await + .unwrap(); + assert_eq!(n, 1); + assert_eq!(state_of(&pool, j1).await, "pending"); + assert_eq!(state_of(&pool, j2).await, "dead"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_skips_dead_when_live_job_exists_for_same_chapter(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "A", 1).await; + let dead = insert_job(&pool, c1, "dead", 5).await; + // A live pending job for the SAME chapter already exists. + insert_job(&pool, c1, "pending", 0).await; + + let n = crawler::requeue_dead_jobs(&pool, RequeueScope::All) + .await + .unwrap(); + assert_eq!(n, 0, "must not resurrect a dead job that has a live counterpart"); + assert_eq!(state_of(&pool, dead).await, "dead"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn requeue_with_two_dead_jobs_for_one_chapter_revives_one_not_500(pool: PgPool) { + // Regression: two dead jobs for the SAME chapter must not both flip to + // pending in one statement — that would violate the partial unique + // dedup index and abort the whole requeue. + let (manga_id, c1) = seed_chapter(&pool, "A", 1).await; + let older = insert_job(&pool, c1, "dead", 5).await; + let newer = insert_job(&pool, c1, "dead", 5).await; + // Make `newer` unambiguously newer. + sqlx::query("UPDATE crawler_jobs SET updated_at = now() - interval '1 hour' WHERE id = $1") + .bind(older) + .execute(&pool) + .await + .unwrap(); + + for scope in [RequeueScope::All, RequeueScope::Manga(manga_id), RequeueScope::Chapter(c1)] { + // Reset to two-dead before each scope variant. + sqlx::query("UPDATE crawler_jobs SET state = 'dead' WHERE id = ANY($1)") + .bind(vec![older, newer]) + .execute(&pool) + .await + .unwrap(); + let n = crawler::requeue_dead_jobs(&pool, scope) + .await + .expect("requeue must not error on duplicate dead jobs"); + assert_eq!(n, 1, "exactly one dead job per chapter is revived"); + // The newest one is the survivor; the other stays dead. + assert_eq!(state_of(&pool, newer).await, "pending"); + assert_eq!(state_of(&pool, older).await, "dead"); + } +} + +#[sqlx::test(migrations = "./migrations")] +async fn list_active_jobs_returns_pending_and_running_running_first(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "Naruto", 700).await; + let (_m2, c2) = seed_chapter(&pool, "Bleach", 10).await; + insert_job(&pool, c1, "pending", 0).await; + insert_job(&pool, c2, "running", 1).await; + // A dead + a done job must NOT appear. + let (_m3, c3) = seed_chapter(&pool, "Gone", 1).await; + insert_job(&pool, c3, "dead", 5).await; + + let (items, total) = crawler::list_active_jobs(&pool, None, 50, 0).await.unwrap(); + assert_eq!(total, 2); + assert_eq!(items.len(), 2); + // Running first. + assert_eq!(items[0].state, "running"); + assert_eq!(items[0].manga_title.as_deref(), Some("Bleach")); + assert_eq!(items[1].state, "pending"); + assert_eq!(items[1].chapter_number, Some(700)); +} + +#[sqlx::test(migrations = "./migrations")] +async fn list_active_jobs_filters_by_title(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "Naruto", 1).await; + let (_m2, c2) = seed_chapter(&pool, "One Piece", 1).await; + insert_job(&pool, c1, "pending", 0).await; + insert_job(&pool, c2, "pending", 0).await; + let (items, total) = crawler::list_active_jobs(&pool, Some("piece"), 50, 0) + .await + .unwrap(); + assert_eq!(total, 1); + assert_eq!(items[0].manga_title.as_deref(), Some("One Piece")); +} + +#[sqlx::test(migrations = "./migrations")] +async fn missing_covers_count_and_list(pool: PgPool) { + seed_missing_cover(&pool, "Naruto").await; + seed_missing_cover(&pool, "Bleach").await; + // A manga WITH a cover must not be counted. + let with_cover = Uuid::new_v4(); + sqlx::query("INSERT INTO mangas (id, title, cover_image_path) VALUES ($1, 'Done', 'k.jpg')") + .bind(with_cover) + .execute(&pool) + .await + .unwrap(); + + assert_eq!(crawler::count_missing_covers(&pool).await.unwrap(), 2); + + let (items, total) = crawler::list_missing_cover_mangas(&pool, None, 50, 0) + .await + .unwrap(); + assert_eq!(total, 2); + assert_eq!(items.len(), 2); + + let (items, total) = crawler::list_missing_cover_mangas(&pool, Some("naru"), 50, 0) + .await + .unwrap(); + assert_eq!(total, 1); + assert_eq!(items[0].manga_title, "Naruto"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn job_state_counts_groups_by_state(pool: PgPool) { + let (_m, c1) = seed_chapter(&pool, "A", 1).await; + let (_m2, c2) = seed_chapter(&pool, "B", 1).await; + let (_m3, c3) = seed_chapter(&pool, "C", 1).await; + insert_job(&pool, c1, "pending", 0).await; + insert_job(&pool, c2, "dead", 5).await; + insert_job(&pool, c3, "dead", 5).await; + + let (pending, running, dead) = crawler::job_state_counts(&pool).await.unwrap(); + assert_eq!(pending, 1); + assert_eq!(running, 0); + assert_eq!(dead, 2); +} diff --git a/backend/tests/crawler_jobs.rs b/backend/tests/crawler_jobs.rs index 7c1fd3b..14a33de 100644 --- a/backend/tests/crawler_jobs.rs +++ b/backend/tests/crawler_jobs.rs @@ -185,6 +185,68 @@ async fn lease_marks_running_and_bumps_attempts_and_sets_leased_until(pool: PgPo assert!(leased_until > chrono::Utc::now()); } +#[sqlx::test(migrations = "./migrations")] +async fn renew_extends_leased_until_while_running(pool: PgPool) { + let id = match jobs::enqueue(&pool, &chapter_content_payload(Uuid::new_v4())) + .await + .unwrap() + { + EnqueueResult::Inserted(id) => id, + EnqueueResult::Skipped => unreachable!(), + }; + + // Lease with a short window, then collapse leased_until to the recent + // past so the renew is unambiguously an extension. + let leases = jobs::lease(&pool, None, 1, Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!(leases.len(), 1); + sqlx::query("UPDATE crawler_jobs SET leased_until = now() - interval '1 second' WHERE id = $1") + .bind(id) + .execute(&pool) + .await + .unwrap(); + + let still_owned = jobs::renew(&pool, id, Duration::from_secs(120)) + .await + .unwrap(); + assert!(still_owned, "renew on a running job returns true"); + + let leased_until: chrono::DateTime = + sqlx::query_scalar("SELECT leased_until FROM crawler_jobs WHERE id = $1") + .bind(id) + .fetch_one(&pool) + .await + .unwrap(); + assert!( + leased_until > chrono::Utc::now() + chrono::Duration::seconds(60), + "leased_until pushed ~120s into the future" + ); + assert_eq!(job_state(&pool, id).await, "running"); +} + +#[sqlx::test(migrations = "./migrations")] +async fn renew_is_noop_once_job_no_longer_running(pool: PgPool) { + let id = match jobs::enqueue(&pool, &chapter_content_payload(Uuid::new_v4())) + .await + .unwrap() + { + EnqueueResult::Inserted(id) => id, + EnqueueResult::Skipped => unreachable!(), + }; + let leases = jobs::lease(&pool, None, 1, Duration::from_secs(60)) + .await + .unwrap(); + // Job completes — heartbeat should now see it's no longer ours. + jobs::ack_done(&pool, leases[0].id).await.unwrap(); + + let still_owned = jobs::renew(&pool, id, Duration::from_secs(120)) + .await + .unwrap(); + assert!(!still_owned, "renew on a non-running job returns false"); + assert_eq!(job_state(&pool, id).await, "done"); +} + #[sqlx::test(migrations = "./migrations")] async fn lease_with_kind_filter_only_matches_that_kind(pool: PgPool) { let manga_id = match jobs::enqueue(&pool, &sync_manga_payload("foo")) diff --git a/backend/tests/repo_chapter.rs b/backend/tests/repo_chapter.rs index 1ed9f05..c2777e8 100644 --- a/backend/tests/repo_chapter.rs +++ b/backend/tests/repo_chapter.rs @@ -109,7 +109,7 @@ async fn dispatch_target_prefers_most_recent_live_source(pool: PgPool) { seed_chapter_with_two_live_sources(&pool).await; let row = dispatch_target(&pool, chapter_id).await.unwrap(); - let (_manga_id, source_url) = + let (_manga_id, source_url, _title, _number) = row.expect("two live sources should yield a dispatch target"); assert_eq!( source_url, new_url, @@ -133,7 +133,7 @@ async fn dispatch_target_skips_dropped_sources(pool: PgPool) { .unwrap(); let row = dispatch_target(&pool, chapter_id).await.unwrap(); - let (_manga_id, source_url) = + let (_manga_id, source_url, _title, _number) = row.expect("a single live source should still yield a dispatch target"); assert!( source_url != new_url, diff --git a/frontend/package.json b/frontend/package.json index f3942ff..4c674e8 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -1,6 +1,6 @@ { "name": "mangalord-frontend", - "version": "0.52.0", + "version": "0.55.0", "private": true, "type": "module", "scripts": { diff --git a/frontend/src/hooks.server.test.ts b/frontend/src/hooks.server.test.ts index 43a4cbd..685fcfd 100644 --- a/frontend/src/hooks.server.test.ts +++ b/frontend/src/hooks.server.test.ts @@ -7,7 +7,7 @@ import { afterEach, type MockInstance } from 'vitest'; -import { handle } from './hooks.server'; +import { handle, shouldBypassProxyTimeout } from './hooks.server'; // `BACKEND_URL` is read at module load time, so the values used in the // asserts below assume the test env didn't set it. `?? 'http://localhost:8080'` @@ -192,3 +192,37 @@ describe('hooks.server proxy', () => { expect(init.signal?.aborted).toBe(false); }); }); + +// ---------------------------------------------------------------------------- +// T1 — SSE bypass for the wall-clock proxy timeout. +// ---------------------------------------------------------------------------- + +describe('shouldBypassProxyTimeout', () => { + it('returns true for an SSE Accept header', () => { + const h = new Headers({ accept: 'text/event-stream' }); + expect(shouldBypassProxyTimeout(h)).toBe(true); + }); + + it('matches case-insensitively', () => { + const h = new Headers({ accept: 'TEXT/EVENT-STREAM' }); + expect(shouldBypassProxyTimeout(h)).toBe(true); + }); + + it('matches when SSE is one of several Accept values', () => { + const h = new Headers({ accept: 'text/event-stream, application/json' }); + expect(shouldBypassProxyTimeout(h)).toBe(true); + }); + + it('returns false for plain JSON / HTML requests', () => { + expect( + shouldBypassProxyTimeout(new Headers({ accept: 'application/json' })) + ).toBe(false); + expect(shouldBypassProxyTimeout(new Headers({ accept: 'text/html' }))).toBe( + false + ); + }); + + it('returns false when Accept is absent', () => { + expect(shouldBypassProxyTimeout(new Headers())).toBe(false); + }); +}); diff --git a/frontend/src/hooks.server.ts b/frontend/src/hooks.server.ts index f413cea..ed34e50 100644 --- a/frontend/src/hooks.server.ts +++ b/frontend/src/hooks.server.ts @@ -46,6 +46,11 @@ const HOP_BY_HOP_HEADERS = [ * tighter upstream proxy may want to lower it. A future improvement * is an idle-based timeout (reset per chunk) instead of this * wall-clock budget — that's a fair bit more code, deferred. + * + * SSE streams (`Accept: text/event-stream`) are exempt — see + * `shouldBypassProxyTimeout`. A wall-clock abort on a long-lived + * stream would tear the connection down every 5 min and force the + * browser to reconnect, which flickers the admin dashboard. */ const PROXY_TIMEOUT_MS = (() => { const raw = process.env.BACKEND_PROXY_TIMEOUT_MS; @@ -53,6 +58,18 @@ const PROXY_TIMEOUT_MS = (() => { return Number.isFinite(n) && n > 0 ? n : 300_000; })(); +/** + * Whether the proxy should skip its wall-clock timeout for this + * request. SSE clients open a single connection and stay subscribed + * for the lifetime of the page; aborting on a wall-clock budget would + * tear down the live admin dashboard every 5 min. Exported for unit + * test coverage. + */ +export function shouldBypassProxyTimeout(headers: Headers): boolean { + const accept = headers.get('accept') ?? ''; + return accept.toLowerCase().includes('text/event-stream'); +} + export const handle: Handle = async ({ event, resolve }) => { if (event.url.pathname.startsWith('/api/')) { const target = `${BACKEND_URL}${event.url.pathname}${event.url.search}`; @@ -63,9 +80,14 @@ export const handle: Handle = async ({ event, resolve }) => { // AbortController times the upstream fetch out so a backend // wedged on a slow DB query doesn't keep the browser request // hanging forever. The `signal` is also wired into the - // RequestInit so the body stream is cancelled cleanly. + // RequestInit so the body stream is cancelled cleanly. For + // SSE streams the timer is suppressed so a long-lived stream + // isn't torn down on the 5-minute mark — see T1 in the audit. + const bypassTimeout = shouldBypassProxyTimeout(event.request.headers); const ctrl = new AbortController(); - const timeoutHandle = setTimeout(() => ctrl.abort(), PROXY_TIMEOUT_MS); + const timeoutHandle = bypassTimeout + ? null + : setTimeout(() => ctrl.abort(), PROXY_TIMEOUT_MS); const init: RequestInit & { duplex?: 'half' } = { method: event.request.method, @@ -91,7 +113,7 @@ export const handle: Handle = async ({ event, resolve }) => { // the real cause. Emit the standard envelope with a // dedicated code instead. console.error('Proxy to backend failed:', e); - clearTimeout(timeoutHandle); + if (timeoutHandle) clearTimeout(timeoutHandle); return new Response( JSON.stringify({ error: { @@ -106,7 +128,7 @@ export const handle: Handle = async ({ event, resolve }) => { ); } - clearTimeout(timeoutHandle); + if (timeoutHandle) clearTimeout(timeoutHandle); return new Response(upstream.body, { status: upstream.status, statusText: upstream.statusText, diff --git a/frontend/src/lib/api/admin.test.ts b/frontend/src/lib/api/admin.test.ts index 01db5ec..3ef4946 100644 --- a/frontend/src/lib/api/admin.test.ts +++ b/frontend/src/lib/api/admin.test.ts @@ -16,7 +16,17 @@ import { listAdminChapters, getSystemStats, resyncManga, - resyncChapter + resyncChapter, + getCrawlerStatus, + crawlerStatusStreamUrl, + runCrawlerPass, + restartCrawlerBrowser, + updateCrawlerSession, + clearCrawlerSessionExpired, + listDeadJobs, + requeueDeadJobs, + listActiveJobs, + listMissingCovers } from './admin'; function ok(body: unknown, status = 200): Response { @@ -329,3 +339,159 @@ describe('admin api client', () => { expect(got.pages).toBeNull(); }); }); + +describe('admin crawler api client', () => { + let fetchSpy: MockInstance; + beforeEach(() => { + fetchSpy = vi.spyOn(globalThis, 'fetch'); + }); + afterEach(() => { + vi.restoreAllMocks(); + }); + + const statusFixture = { + daemon: 'running', + phase: { state: 'fetching_metadata', index: 3, total: 10, title: 'One Piece' }, + worker_count: 2, + active_chapters: [ + { + manga_id: 'm-1', + manga_title: 'Bleach', + chapter_id: 'c-1', + chapter_number: 12, + pages_done: 4, + pages_total: 20 + } + ], + current_cover: { manga_id: 'm-2', manga_title: 'Naruto' }, + covers_queued: 7, + last_pass: { at: null, discovered: 0, upserted: 0, covers_fetched: 0, mangas_failed: 0 }, + session: { expired: false, configured: true }, + browser: 'healthy', + queue: { pending: 2, running: 1, dead: 4 } + }; + + it('crawlerStatusStreamUrl points at the SSE endpoint under the API base', () => { + expect(crawlerStatusStreamUrl()).toMatch(/\/v1\/admin\/crawler\/stream$/); + }); + + it('getCrawlerStatus GETs /v1/admin/crawler with live chapter/cover fields', async () => { + fetchSpy.mockResolvedValueOnce(ok(statusFixture)); + const s = await getCrawlerStatus(); + expect(s.queue.dead).toBe(4); + expect(s.phase?.state).toBe('fetching_metadata'); + expect(s.active_chapters[0].pages_done).toBe(4); + expect(s.active_chapters[0].pages_total).toBe(20); + expect(s.current_cover?.manga_title).toBe('Naruto'); + expect(s.covers_queued).toBe(7); + const url = fetchSpy.mock.calls[0][0] as string; + expect(url).toMatch(/\/v1\/admin\/crawler$/); + }); + + it('listActiveJobs GETs /v1/admin/crawler/active-jobs with search', async () => { + fetchSpy.mockResolvedValueOnce( + ok({ items: [], page: { limit: 20, offset: 0, total: 0 } }) + ); + await listActiveJobs({ search: 'bleach' }); + const url = fetchSpy.mock.calls[0][0] as string; + expect(url).toMatch(/\/v1\/admin\/crawler\/active-jobs\?/); + expect(url).toContain('search=bleach'); + }); + + it('listMissingCovers GETs /v1/admin/crawler/covers', async () => { + fetchSpy.mockResolvedValueOnce( + ok({ items: [{ manga_id: 'm-1', manga_title: 'X' }], page: { limit: 20, offset: 0, total: 1 } }) + ); + const r = await listMissingCovers(); + expect(r.items[0].manga_title).toBe('X'); + expect(fetchSpy.mock.calls[0][0]).toMatch(/\/v1\/admin\/crawler\/covers$/); + }); + + it('runCrawlerPass POSTs /v1/admin/crawler/run', async () => { + fetchSpy.mockResolvedValueOnce(ok({ started: true })); + const r = await runCrawlerPass(); + expect(r.started).toBe(true); + const init = fetchSpy.mock.calls[0][1] as RequestInit; + expect(init.method).toBe('POST'); + expect(fetchSpy.mock.calls[0][0]).toMatch(/\/v1\/admin\/crawler\/run$/); + }); + + it('restartCrawlerBrowser POSTs the restart endpoint', async () => { + fetchSpy.mockResolvedValueOnce(ok({ ok: true, error: null })); + const r = await restartCrawlerBrowser(); + expect(r.ok).toBe(true); + expect(fetchSpy.mock.calls[0][0]).toMatch(/\/v1\/admin\/crawler\/browser\/restart$/); + }); + + it('updateCrawlerSession POSTs the phpsessid body', async () => { + fetchSpy.mockResolvedValueOnce(ok({ valid: true, error: null })); + const r = await updateCrawlerSession('abc123'); + expect(r.valid).toBe(true); + const init = fetchSpy.mock.calls[0][1] as RequestInit; + expect(init.method).toBe('POST'); + expect(JSON.parse(init.body as string)).toEqual({ phpsessid: 'abc123' }); + }); + + it('clearCrawlerSessionExpired POSTs clear-expired', async () => { + fetchSpy.mockResolvedValueOnce(ok({ cleared: true })); + const r = await clearCrawlerSessionExpired(); + expect(r.cleared).toBe(true); + expect(fetchSpy.mock.calls[0][0]).toMatch(/\/v1\/admin\/crawler\/session\/clear-expired$/); + }); + + it('listDeadJobs forwards search + pagination', async () => { + fetchSpy.mockResolvedValueOnce( + ok({ items: [], page: { limit: 20, offset: 20, total: 0 } }) + ); + await listDeadJobs({ search: 'naruto', limit: 20, offset: 20 }); + const url = fetchSpy.mock.calls[0][0] as string; + expect(url).toContain('search=naruto'); + expect(url).toContain('offset=20'); + }); + + it('requeueDeadJobs POSTs the scope body', async () => { + fetchSpy.mockResolvedValueOnce(ok({ requeued: 3 })); + const r = await requeueDeadJobs({ scope: 'manga', manga_id: 'm-9' }); + expect(r.requeued).toBe(3); + const init = fetchSpy.mock.calls[0][1] as RequestInit; + expect(JSON.parse(init.body as string)).toEqual({ scope: 'manga', manga_id: 'm-9' }); + }); + + it('requeueDeadJobs serialises chapter scope verbatim', async () => { + // F7 — the per-chapter requeue button in /admin/mangas sends this + // exact shape; pin it so a future rename of `chapter_id` doesn't + // silently break the inline button. + fetchSpy.mockResolvedValueOnce(ok({ requeued: 1 })); + const r = await requeueDeadJobs({ scope: 'chapter', chapter_id: 'c-7' }); + expect(r.requeued).toBe(1); + const init = fetchSpy.mock.calls[0][1] as RequestInit; + expect(JSON.parse(init.body as string)).toEqual({ + scope: 'chapter', + chapter_id: 'c-7' + }); + }); + + it('requeueDeadJobs serialises job + all (with confirm) scopes', async () => { + // Round out the variant coverage: { scope: 'job', job_id } and + // { scope: 'all', confirm: true }. The audit (S1) requires the + // confirm flag on the wire for scope=all; this pins it so a + // future refactor can't drop it. + fetchSpy.mockResolvedValueOnce(ok({ requeued: 1 })); + await requeueDeadJobs({ scope: 'job', job_id: 'j-1' }); + expect(JSON.parse(fetchSpy.mock.calls[0][1]!.body as string)).toEqual({ + scope: 'job', + job_id: 'j-1' + }); + fetchSpy.mockResolvedValueOnce(ok({ requeued: 99 })); + await requeueDeadJobs({ scope: 'all', confirm: true }); + expect(JSON.parse(fetchSpy.mock.calls[1][1]!.body as string)).toEqual({ + scope: 'all', + confirm: true + }); + }); + + it('surfaces a 503 as ApiError', async () => { + fetchSpy.mockResolvedValueOnce(envelope(503, 'service_unavailable', 'disabled')); + await expect(runCrawlerPass()).rejects.toMatchObject({ status: 503 }); + }); +}); diff --git a/frontend/src/lib/api/admin.ts b/frontend/src/lib/api/admin.ts index 930342d..547e622 100644 --- a/frontend/src/lib/api/admin.ts +++ b/frontend/src/lib/api/admin.ts @@ -3,7 +3,7 @@ // won't reach these routes). 403s thrown here propagate up to the // /admin layout, which renders the framework error page. -import { request, type Page } from './client'; +import { request, apiUrl, type Page } from './client'; import type { User } from './auth'; import type { MangaDetail } from './mangas'; import type { Chapter } from './chapters'; @@ -214,3 +214,192 @@ export async function resyncChapter(id: string): Promise { method: 'POST' } ); } + +// ---- crawler observability + control --------------------------------------- + +/** Current daemon activity. Discriminated on `state`. */ +export type CrawlerPhase = + | { state: 'idle'; next_fire: string | null } + | { state: 'walking_list' } + | { state: 'fetching_metadata'; index: number; total: number | null; title: string } + | { state: 'cover_backfill'; index: number; total: number }; + +/** A chapter being crawled right now, with a live page count. */ +export type ActiveChapter = { + manga_id: string; + manga_title: string; + chapter_id: string; + chapter_number: number; + pages_done: number; + pages_total: number | null; +}; + +export type CrawlerLastPass = { + at: string | null; + discovered: number; + upserted: number; + covers_fetched: number; + mangas_failed: number; +}; + +export type CrawlerStatus = { + daemon: 'running' | 'disabled'; + phase: CrawlerPhase | null; + worker_count: number; + active_chapters: ActiveChapter[]; + current_cover: { manga_id: string; manga_title: string } | null; + covers_queued: number; + last_pass: CrawlerLastPass; + session: { expired: boolean; configured: boolean }; + browser: 'healthy' | 'draining' | 'restarting' | 'down'; + queue: { pending: number; running: number; dead: number }; +}; + +export async function getCrawlerStatus(): Promise { + return request('/v1/admin/crawler'); +} + +/** URL of the Server-Sent Events live-status stream. Open with + * `new EventSource(...)` while the crawler page is mounted and close it on + * navigate-away so the subscription is scoped to the active page. Each + * message is a named `status` event whose `data` is a {@link CrawlerStatus}. */ +export function crawlerStatusStreamUrl(): string { + return apiUrl('/v1/admin/crawler/stream'); +} + +/** POST /v1/admin/crawler/run — trigger an out-of-cycle metadata pass. */ +export async function runCrawlerPass(): Promise<{ started: boolean }> { + return request('/v1/admin/crawler/run', { method: 'POST' }); +} + +/** POST /v1/admin/crawler/browser/restart — coordinated Chromium restart. */ +export async function restartCrawlerBrowser(): Promise<{ ok: boolean; error: string | null }> { + return request('/v1/admin/crawler/browser/restart', { method: 'POST' }); +} + +/** POST /v1/admin/crawler/session — refresh PHPSESSID and re-probe. */ +export async function updateCrawlerSession( + phpsessid: string +): Promise<{ valid: boolean; error: string | null }> { + return request('/v1/admin/crawler/session', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ phpsessid }) + }); +} + +/** POST /v1/admin/crawler/session/clear-expired — resume idled workers. */ +export async function clearCrawlerSessionExpired(): Promise<{ cleared: boolean }> { + return request('/v1/admin/crawler/session/clear-expired', { method: 'POST' }); +} + +export type DeadJob = { + id: string; + kind: string; + chapter_id: string | null; + manga_id: string | null; + manga_title: string | null; + chapter_number: number | null; + attempts: number; + max_attempts: number; + last_error: string | null; + updated_at: string; +}; + +export type DeadJobsPage = { items: DeadJob[]; page: Page }; + +export async function listDeadJobs( + opts?: { + search?: string; + limit?: number; + offset?: number; + }, + init?: RequestInit +): Promise { + const params = new URLSearchParams(); + if (opts?.search) params.set('search', opts.search); + if (opts?.limit != null) params.set('limit', String(opts.limit)); + if (opts?.offset != null) params.set('offset', String(opts.offset)); + const qs = params.toString(); + return request( + `/v1/admin/crawler/dead-jobs${qs ? `?${qs}` : ''}`, + init + ); +} + +/** Requeue scope: all dead jobs, one manga's, one chapter's, or a single job. + * `scope: 'all'` requires an explicit `confirm: true` so a careless + * click (or CSRF bait) can't flip the entire dead pile. */ +export type RequeueScope = + | { scope: 'all'; confirm: true } + | { scope: 'manga'; manga_id: string } + | { scope: 'chapter'; chapter_id: string } + | { scope: 'job'; job_id: string }; + +export async function requeueDeadJobs(scope: RequeueScope): Promise<{ requeued: number }> { + return request('/v1/admin/crawler/dead-jobs/requeue', { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify(scope) + }); +} + +/** A queued/running chapter-content job (which chapters are queued). */ +export type ActiveJob = { + id: string; + chapter_id: string | null; + manga_id: string | null; + manga_title: string | null; + chapter_number: number | null; + state: 'pending' | 'running'; + attempts: number; + max_attempts: number; + updated_at: string; +}; + +export type ActiveJobsPage = { items: ActiveJob[]; page: Page }; + +/** GET /v1/admin/crawler/active-jobs — which chapters of which mangas are + * queued or running now. */ +export async function listActiveJobs( + opts?: { + search?: string; + limit?: number; + offset?: number; + }, + init?: RequestInit +): Promise { + const params = new URLSearchParams(); + if (opts?.search) params.set('search', opts.search); + if (opts?.limit != null) params.set('limit', String(opts.limit)); + if (opts?.offset != null) params.set('offset', String(opts.offset)); + const qs = params.toString(); + return request( + `/v1/admin/crawler/active-jobs${qs ? `?${qs}` : ''}`, + init + ); +} + +/** A manga queued for a cover fetch (no cover yet + a live source). */ +export type MissingCover = { manga_id: string; manga_title: string }; +export type MissingCoversPage = { items: MissingCover[]; page: Page }; + +/** GET /v1/admin/crawler/covers — which manga covers are queued. */ +export async function listMissingCovers( + opts?: { + search?: string; + limit?: number; + offset?: number; + }, + init?: RequestInit +): Promise { + const params = new URLSearchParams(); + if (opts?.search) params.set('search', opts.search); + if (opts?.limit != null) params.set('limit', String(opts.limit)); + if (opts?.offset != null) params.set('offset', String(opts.offset)); + const qs = params.toString(); + return request( + `/v1/admin/crawler/covers${qs ? `?${qs}` : ''}`, + init + ); +} diff --git a/frontend/src/lib/api/client.ts b/frontend/src/lib/api/client.ts index 6b690be..c96414f 100644 --- a/frontend/src/lib/api/client.ts +++ b/frontend/src/lib/api/client.ts @@ -12,6 +12,15 @@ export function fileUrl(key: string): string { return `${BASE}/v1/files/${key}`; } +/** + * Builds an API URL for non-`fetch` consumers (e.g. `EventSource` for SSE), + * applying the same `VITE_API_BASE` prefix as `request()`. `path` is the + * route after the base, e.g. `/v1/admin/crawler/stream`. + */ +export function apiUrl(path: string): string { + return `${BASE}${path}`; +} + export class ApiError extends Error { constructor( public readonly status: number, diff --git a/frontend/src/lib/components/crawler/ActiveChaptersCard.svelte b/frontend/src/lib/components/crawler/ActiveChaptersCard.svelte new file mode 100644 index 0000000..d2579a0 --- /dev/null +++ b/frontend/src/lib/components/crawler/ActiveChaptersCard.svelte @@ -0,0 +1,112 @@ + + +
+
+

Queue

+
+
Pending
+
{status.queue.pending}
+
Running
+
{status.queue.running}
+
Dead
+
{status.queue.dead}
+
Covers queued
+
{status.covers_queued}
+
+
+
+

Active chapters ({status.active_chapters.length}/{status.worker_count})

+ {#if status.active_chapters.length === 0} +

idle — no chapters downloading

+ {:else} + + + {#each status.active_chapters as c (c.chapter_id)} + + + + + + {/each} + +
{c.manga_title} · ch.{c.chapter_number} + {c.pages_done}/{c.pages_total ?? '?'} + + {#if chapterPercent(c) !== null} + + {/if} +
+ {/if} +
+
+ + diff --git a/frontend/src/lib/components/crawler/ActiveJobsTable.svelte b/frontend/src/lib/components/crawler/ActiveJobsTable.svelte new file mode 100644 index 0000000..bbbe268 --- /dev/null +++ b/frontend/src/lib/components/crawler/ActiveJobsTable.svelte @@ -0,0 +1,101 @@ + + +
+
+

Queued chapters ({total})

+
+ +
+
+ {#if jobs.length === 0} +

No chapters queued.

+ {:else} + + + + + + + + + + {#each jobs as j (j.id)} + + + + + + {/each} + +
Manga / ChapterStateAtt.
+ {j.manga_title ?? '(unknown)'} + {#if j.chapter_number != null}· ch.{j.chapter_number}{/if} + + {j.state} + {j.attempts}/{j.max_attempts}
+ + {/if} +
+ + diff --git a/frontend/src/lib/components/crawler/CrawlerControls.svelte b/frontend/src/lib/components/crawler/CrawlerControls.svelte new file mode 100644 index 0000000..a66b8af --- /dev/null +++ b/frontend/src/lib/components/crawler/CrawlerControls.svelte @@ -0,0 +1,43 @@ + + +
+ + + + {#if status.session.expired} + + {/if} +
+ + diff --git a/frontend/src/lib/components/crawler/CrawlerHero.svelte b/frontend/src/lib/components/crawler/CrawlerHero.svelte new file mode 100644 index 0000000..be27492 --- /dev/null +++ b/frontend/src/lib/components/crawler/CrawlerHero.svelte @@ -0,0 +1,148 @@ + + +
+
+ Daemon + {status.daemon} + Session + {sessionPill(status).text} + Browser + {browserPill(status).text} +
+ +

{phaseLabel(status.phase)}

+ {#if phasePercent(status.phase) !== null} + + {/if} + + {#if status.session.expired} +

+ ⚠ Chapter downloads paused — session expired. Metadata + list crawl continue. +

+ {/if} + + {#if status.current_cover} +

+ 🖼 Fetching cover: {status.current_cover.manga_title} +

+ {/if} + +

+ Last pass: + {#if status.last_pass.at} + {new Date(status.last_pass.at).toLocaleString()} · + {status.last_pass.discovered} seen · {status.last_pass.upserted} upserted · + {status.last_pass.mangas_failed} failed + {:else} + — none yet this session + {/if} +

+
+ + diff --git a/frontend/src/lib/components/crawler/DeadJobsTable.svelte b/frontend/src/lib/components/crawler/DeadJobsTable.svelte new file mode 100644 index 0000000..2de8210 --- /dev/null +++ b/frontend/src/lib/components/crawler/DeadJobsTable.svelte @@ -0,0 +1,132 @@ + + +
+
+

Dead jobs ({total})

+
+ + +
+
+ + {#if jobs.length === 0} +

No dead jobs 🎉

+ {:else} + + + + + + + + + + + + {#each jobs as j (j.id)} + + + + + + + + {/each} + +
Manga / ChapterAtt.FailedLast errorAction
+ {j.manga_title ?? '(unknown)'} + {#if j.chapter_number != null}· ch.{j.chapter_number}{/if} + {j.attempts}/{j.max_attempts}{new Date(j.updated_at).toLocaleDateString()}{j.last_error ?? '—'} + + {#if j.manga_id} + + {/if} +
+ + {/if} +
+ + diff --git a/frontend/src/lib/components/crawler/MissingCoversTable.svelte b/frontend/src/lib/components/crawler/MissingCoversTable.svelte new file mode 100644 index 0000000..8165595 --- /dev/null +++ b/frontend/src/lib/components/crawler/MissingCoversTable.svelte @@ -0,0 +1,85 @@ + + +
+
+

Queued covers ({total})

+
+ +
+
+ {#if covers.length === 0} +

No covers queued 🎉

+ {:else} + + + + + + {#each covers as c (c.manga_id)} + + {/each} + +
Manga
{c.manga_title}
+ + {/if} +
+ + diff --git a/frontend/src/lib/components/crawler/ProgressBar.svelte b/frontend/src/lib/components/crawler/ProgressBar.svelte new file mode 100644 index 0000000..c8deab8 --- /dev/null +++ b/frontend/src/lib/components/crawler/ProgressBar.svelte @@ -0,0 +1,35 @@ + + +
+
+ {percent.toFixed(0)}% +
+ + diff --git a/frontend/src/lib/components/crawler/RequeueAllConfirmModal.svelte b/frontend/src/lib/components/crawler/RequeueAllConfirmModal.svelte new file mode 100644 index 0000000..479d87b --- /dev/null +++ b/frontend/src/lib/components/crawler/RequeueAllConfirmModal.svelte @@ -0,0 +1,32 @@ + + + + {#snippet children()} +

+ This will flip every dead job ({total}) back to pending. + Chapters with a live job are skipped automatically; the rest will be + picked up by the next chapter worker. +

+ {/snippet} + {#snippet footer()} + + + + {/snippet} +
diff --git a/frontend/src/lib/components/crawler/RestartConfirmModal.svelte b/frontend/src/lib/components/crawler/RestartConfirmModal.svelte new file mode 100644 index 0000000..3c11a6b --- /dev/null +++ b/frontend/src/lib/components/crawler/RestartConfirmModal.svelte @@ -0,0 +1,42 @@ + + + + {#snippet children()} +

This relaunches Chromium and re-injects the session cookie.

+
    +
  • In-flight jobs are allowed to finish (bounded), then forced.
  • +
  • New jobs pause until the relaunch completes.
  • +
  • The metadata pass yields at its next checkpoint.
  • +
+ {/snippet} + {#snippet footer()} + + + + + {/snippet} +
+ + diff --git a/frontend/src/lib/components/crawler/SearchBar.svelte b/frontend/src/lib/components/crawler/SearchBar.svelte new file mode 100644 index 0000000..67e2032 --- /dev/null +++ b/frontend/src/lib/components/crawler/SearchBar.svelte @@ -0,0 +1,22 @@ + + + e.key === 'Enter' && onSearch()} +/> + diff --git a/frontend/src/lib/components/crawler/SessionModal.svelte b/frontend/src/lib/components/crawler/SessionModal.svelte new file mode 100644 index 0000000..86708f7 --- /dev/null +++ b/frontend/src/lib/components/crawler/SessionModal.svelte @@ -0,0 +1,49 @@ + + + + {#snippet children()} + + +

+ Saving rewrites the cookie everywhere, persists it, restarts the browser, and re-probes. +

+ {#if result} +

{result}

+ {/if} + {/snippet} + {#snippet footer()} + + + {/snippet} +
+ + diff --git a/frontend/src/routes/admin/+layout.svelte b/frontend/src/routes/admin/+layout.svelte index 58575da..1da9af1 100644 --- a/frontend/src/routes/admin/+layout.svelte +++ b/frontend/src/routes/admin/+layout.svelte @@ -6,6 +6,7 @@ { href: '/admin', label: 'Overview' }, { href: '/admin/users', label: 'Users' }, { href: '/admin/mangas', label: 'Mangas' }, + { href: '/admin/crawler', label: 'Crawler' }, { href: '/admin/system', label: 'System' } ]; diff --git a/frontend/src/routes/admin/crawler/+page.svelte b/frontend/src/routes/admin/crawler/+page.svelte new file mode 100644 index 0000000..faa1213 --- /dev/null +++ b/frontend/src/routes/admin/crawler/+page.svelte @@ -0,0 +1,532 @@ + + +
+

Crawler

+ + {live ? '● live' : '○ reconnecting…'} + +
+ +{#if error} + +{/if} +{#if notice} +

{notice}

+{/if} + +{#if status} + + + (restartModalOpen = true)} + onOpenSession={() => { + sessionModalOpen = true; + sessionResult = null; + }} + {onClearExpired} + /> + + +{:else} +

Loading…

+{/if} + + + + + + (requeueAllModalOpen = true)} +/> + + (restartModalOpen = false)} + onConfirm={onConfirmRestart} +/> + + (requeueAllModalOpen = false)} + onConfirm={onConfirmRequeueAll} +/> + + (sessionModalOpen = false)} + onSave={onSaveSession} +/> + + diff --git a/frontend/src/routes/admin/mangas/+page.svelte b/frontend/src/routes/admin/mangas/+page.svelte index 3824585..36a85a4 100644 --- a/frontend/src/routes/admin/mangas/+page.svelte +++ b/frontend/src/routes/admin/mangas/+page.svelte @@ -3,6 +3,7 @@ import { listAdminMangas, listAdminChapters, + requeueDeadJobs, type AdminMangasPage, type AdminChapterRow, type MangaSyncState @@ -59,6 +60,39 @@ function badgeClass(state: string): string { return `badge badge-${state}`; } + + let requeuingChapter: string | null = $state(null); + + /** Requeue the dead job(s) for a single failed chapter, then patch + * the local state instead of refetching the whole list. Refetching + * 500 chapters every time an operator clicks "Requeue" was wasteful; + * the only field that changes is the per-chapter sync state of the + * affected row. F4 in the audit. */ + async function requeueChapter(mangaId: string, chapterId: string) { + requeuingChapter = chapterId; + error = null; + try { + await requeueDeadJobs({ scope: 'chapter', chapter_id: chapterId }); + const existing = chaptersByManga[mangaId]; + if (existing && existing !== 'loading') { + // After requeue the dead job is back to `pending`; the + // chapter itself stays "not_downloaded" until a worker + // picks it up and stores pages. + chaptersByManga[mangaId] = { + items: existing.items.map((c) => + c.id === chapterId + ? { ...c, sync_state: 'not_downloaded' as const } + : c + ), + total: existing.total + }; + } + } catch (e) { + error = e instanceof ApiError ? e.message : 'requeue failed'; + } finally { + requeuingChapter = null; + } + }

Mangas

@@ -153,6 +187,17 @@ {c.sync_state} + {#if c.sync_state === 'failed'} + + {/if} {/each} @@ -272,6 +317,11 @@ color: #991b1b; border-color: #fca5a5; } + .requeue { + margin-left: var(--space-2); + font-size: var(--font-xs); + padding: 0 var(--space-2); + } .badge-not_downloaded { background: var(--surface-elevated); color: var(--text-muted);