feat(crawler): reconcile pass to enqueue mangas missing from the DB (0.85.0)
The interleaved metadata pass misses mangas to list drift (a title slips a pagination slot during the slow detail walk). Add a reconcile pass: a cheap, full, list-only walk (refs only, no detail visit, no early stop) that set-diffs the walked keys against manga_sources and enqueues the strictly- missing ones as SyncManga jobs. A set-diff is immune to drift — a manga only has to appear somewhere in the list. - Build the previously-dead SyncManga worker by extending RealChapterDispatcher to run the shared pipeline::process_manga_ref (fetch → upsert → cover → chapters), refactored out of run_metadata_pass so both paths stay in lockstep. The crawl worker now leases both sync_chapter_content and sync_manga (jobs::lease_kinds); both serialize on the single exclusive browser. - SyncManga payload carries url + title so the worker can rebuild the ref and the dead-jobs/history UI can label a missing manga that has no manga row yet. - "Missing" = strict NOT EXISTS (dropped rows count as present). Re-enqueue is skipped when a pending/running/dead SyncManga job already exists, so a gone manga (detail 404 → retries → dead) is left dead and not retried. - New POST /v1/admin/crawler/reconcile (fire-and-forget, shares manual_pass_lock) + "Reconcile missing" admin button + Reconciling status phase streamed over SSE. - dead-jobs/history queries surface payload title/url/key; tables fall back to them for sync_manga rows; both searches match the payload title. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -116,6 +116,57 @@ async fn list_dead_jobs_filters_by_title_search(pool: PgPool) {
|
||||
assert_eq!(items[0].manga_title.as_deref(), Some("One Piece"));
|
||||
}
|
||||
|
||||
/// Insert a dead `sync_manga` job (no manga/chapter rows) — the reconcile
|
||||
/// "detail page gone" case.
|
||||
async fn insert_dead_sync_manga(pool: &PgPool, key: &str, title: &str) -> Uuid {
|
||||
let id = Uuid::new_v4();
|
||||
let payload = json!({
|
||||
"kind": "sync_manga",
|
||||
"source_id": "target",
|
||||
"source_manga_key": key,
|
||||
"url": format!("http://x/manga/{key}"),
|
||||
"title": title,
|
||||
});
|
||||
sqlx::query(
|
||||
"INSERT INTO crawler_jobs (id, payload, state, attempts, last_error) \
|
||||
VALUES ($1, $2, 'dead', 5, 'broken-page body signature')",
|
||||
)
|
||||
.bind(id)
|
||||
.bind(payload)
|
||||
.execute(pool)
|
||||
.await
|
||||
.unwrap();
|
||||
id
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dead_sync_manga_job_surfaces_payload_title_and_url(pool: PgPool) {
|
||||
insert_dead_sync_manga(&pool, "gone-1", "Ghost Manga").await;
|
||||
|
||||
let (items, total) = crawler::list_dead_jobs(&pool, None, 50, 0).await.unwrap();
|
||||
assert_eq!(total, 1);
|
||||
let row = &items[0];
|
||||
// No manga row exists → manga_title is None; the payload fields fill in.
|
||||
assert_eq!(row.manga_title, None);
|
||||
assert_eq!(row.payload_title.as_deref(), Some("Ghost Manga"));
|
||||
assert_eq!(row.source_url.as_deref(), Some("http://x/manga/gone-1"));
|
||||
assert_eq!(row.source_key.as_deref(), Some("gone-1"));
|
||||
assert_eq!(row.kind, "sync_manga");
|
||||
assert_eq!(row.last_error.as_deref(), Some("broken-page body signature"));
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn dead_jobs_search_matches_payload_title(pool: PgPool) {
|
||||
insert_dead_sync_manga(&pool, "gone-1", "Ghost Manga").await;
|
||||
insert_dead_sync_manga(&pool, "gone-2", "Other Title").await;
|
||||
|
||||
let (items, total) = crawler::list_dead_jobs(&pool, Some("ghost"), 50, 0)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(total, 1);
|
||||
assert_eq!(items[0].payload_title.as_deref(), Some("Ghost Manga"));
|
||||
}
|
||||
|
||||
#[sqlx::test(migrations = "./migrations")]
|
||||
async fn requeue_all_resets_dead_jobs_to_pending(pool: PgPool) {
|
||||
let (_m, c1) = seed_chapter(&pool, "A", 1).await;
|
||||
|
||||
Reference in New Issue
Block a user