fix: batch and index the crawler_jobs retention reaper
reap_terminal was one unbatched DELETE with no updated_at index — a full scan and
long-held lock over a large backlog (audit M5). Migration 0040 adds a partial
index on (updated_at) WHERE state IN ('done','dead'); reap_terminal now deletes in
bounded 5k batches. Existing tests green.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2
backend/Cargo.lock
generated
2
backend/Cargo.lock
generated
@@ -1558,7 +1558,7 @@ checksum = "c41e0c4fef86961ac6d6f8a82609f55f31b05e4fce149ac5710e439df7619ba4"
|
||||
|
||||
[[package]]
|
||||
name = "mangalord"
|
||||
version = "0.128.19"
|
||||
version = "0.128.20"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"argon2",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "mangalord"
|
||||
version = "0.128.19"
|
||||
version = "0.128.20"
|
||||
edition = "2021"
|
||||
default-run = "mangalord"
|
||||
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
-- Index the retention reaper's scan. reap_terminal deletes
|
||||
-- WHERE state IN ('done','dead') AND updated_at < now() - interval
|
||||
-- which had no supporting index, so each daily sweep sequential-scanned the whole
|
||||
-- crawler_jobs table to find expired terminal rows (audit M5). A partial index on
|
||||
-- updated_at over just the terminal states makes the batched reaper's per-batch
|
||||
-- SELECT index-backed.
|
||||
CREATE INDEX crawler_jobs_terminal_reap_idx
|
||||
ON crawler_jobs (updated_at)
|
||||
WHERE state IN ('done', 'dead');
|
||||
@@ -495,15 +495,33 @@ pub async fn reap_terminal(pool: &PgPool, retention_days: u32) -> sqlx::Result<u
|
||||
if retention_days == 0 {
|
||||
return Ok(0);
|
||||
}
|
||||
let result = sqlx::query(
|
||||
"DELETE FROM crawler_jobs \
|
||||
WHERE state IN ('done', 'dead') \
|
||||
AND updated_at < now() - ($1::bigint || ' days')::interval",
|
||||
)
|
||||
.bind(retention_days as i64)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
Ok(result.rows_affected())
|
||||
// Delete in bounded batches rather than one statement. A single unbatched
|
||||
// DELETE over an unbounded backlog takes a long-held lock and one huge
|
||||
// transaction; batching keeps each delete short (index-backed by
|
||||
// crawler_jobs_terminal_reap_idx, 0040) so the reaper never pins the table
|
||||
// (audit M5). Loop until a batch comes back short.
|
||||
const BATCH: i64 = 5_000;
|
||||
let mut total: u64 = 0;
|
||||
loop {
|
||||
let result = sqlx::query(
|
||||
"DELETE FROM crawler_jobs \
|
||||
WHERE ctid IN ( \
|
||||
SELECT ctid FROM crawler_jobs \
|
||||
WHERE state IN ('done', 'dead') \
|
||||
AND updated_at < now() - ($1::bigint || ' days')::interval \
|
||||
LIMIT $2 )",
|
||||
)
|
||||
.bind(retention_days as i64)
|
||||
.bind(BATCH)
|
||||
.execute(pool)
|
||||
.await?;
|
||||
let n = result.rows_affected();
|
||||
total += n;
|
||||
if n < BATCH as u64 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Ok(total)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
Reference in New Issue
Block a user