40 lines
2.2 KiB
SQL
40 lines
2.2 KiB
SQL
-- Durable timing log for crawler operations + a duration column for page
|
|
-- analysis. The dashboard's job history shows *what* ran; this records *how
|
|
-- long it took* so operators can see per-operation durations and averages
|
|
-- broken down by type/granularity (manga list walk, manga detail, cover,
|
|
-- whole chapter — per-page crawl timing is derived from the chapter row's
|
|
-- `items` count rather than stored per image).
|
|
--
|
|
-- This table is durable on purpose: it outlives the `crawler_jobs` done-job
|
|
-- reaper so averages stay meaningful, and it captures the INLINE operations
|
|
-- (list walk, manga detail, cover) that are not queue jobs at all. Volume is
|
|
-- low (one row per manga / chapter / pass, not per image), so a generous
|
|
-- retention reaper is hygiene rather than a necessity.
|
|
CREATE TABLE crawl_metrics (
|
|
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
|
-- Operation granularity. analyze_page is intentionally absent — analysis
|
|
-- duration lives on page_analysis.duration_ms (one row per page already).
|
|
op text NOT NULL
|
|
CHECK (op IN ('manga_list', 'manga_detail', 'manga_cover', 'chapter')),
|
|
-- Best-effort target context for labeling/drill. SET NULL so deleting a
|
|
-- manga/chapter doesn't erase its historical timings.
|
|
manga_id uuid REFERENCES mangas(id) ON DELETE SET NULL,
|
|
chapter_id uuid REFERENCES chapters(id) ON DELETE SET NULL,
|
|
outcome text NOT NULL CHECK (outcome IN ('ok', 'failed')),
|
|
duration_ms bigint NOT NULL,
|
|
-- Unit count for the op: chapter = pages stored; manga_list = mangas
|
|
-- discovered. Drives the derived "per page" average. NULL when N/A.
|
|
items integer,
|
|
error text,
|
|
finished_at timestamptz NOT NULL DEFAULT now()
|
|
);
|
|
|
|
-- Per-type averages + the recent-ops log both filter by op and order by time.
|
|
CREATE INDEX crawl_metrics_op_time_idx ON crawl_metrics (op, finished_at DESC);
|
|
-- The window filter (last 24h / 7d / 30d) and the retention reaper scan by time.
|
|
CREATE INDEX crawl_metrics_time_idx ON crawl_metrics (finished_at);
|
|
|
|
-- Wall-clock the analysis worker spent on a page (vision dispatch). NULL for
|
|
-- pre-existing rows and any page analyzed before this migration.
|
|
ALTER TABLE page_analysis ADD COLUMN duration_ms bigint;
|