chore: dedupe is_unique_violation, lift SQL into repo, centralise URL parsing

Three layering cleanups from REVIEW.md §5 / §3: - Drop the three private `is_unique_violation` helpers in repo::{user,chapter,bookmark} in favour of sqlx 0.8's `DatabaseError::is_unique_violation()` method (already used by repo::collection). - Remove the unreachable 23505 branch in repo::chapter::create — the (manga_id, number) UNIQUE was dropped in 0013, so the defensive arm could no longer fire. A doc note records what to do if uniqueness is re-added. - Move three inline SQL queries out of handlers/daemon into repo functions: bookmarks' chapter-belongs-to-manga guard (`repo::chapter::belongs_to_manga`), the daemon's dispatch lookup (`repo::chapter::dispatch_target`), and the daemon's page_count safety net (`repo::chapter::page_count`). Restores the handlers→repo layering invariant in CLAUDE.md. - New `crawler::url_utils` module consolidates host_of / origin_of / registrable_domain — they used to live in three crawler submodules with diverging edge-case behaviour. Tests moved with them. - Doc cross-references on repo::author::set_for_manga and repo::genre::set_for_manga pointing to the crawler's name-keyed variants, so the intentional duplication is discoverable. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-28 19:31:49 +02:00
parent bd9a6bd257
commit c320eda7cd
13 changed files with 282 additions and 149 deletions
--- a/backend/src/crawler/session.rs
+++ b/backend/src/crawler/session.rs
@@ -42,36 +42,9 @@ pub enum SessionProbe {
    Transient,
 }

-/// Compute the cookie domain (e.g. `.example.com`) from a start URL.
-/// The leading dot makes the cookie cover every subdomain — the source
-/// often redirects between `www.` and other prefixes mid-crawl, and a
-/// host-only cookie would silently drop on the cross-subdomain hop.
-///
-/// Caveat: this takes the last two dot-labels, which is wrong for
-/// multi-part TLDs (`.co.uk`, `.com.br` would resolve to `.co.uk` and
-/// attach to every site on `.co.uk`). For those, the operator should
-/// override via `CRAWLER_COOKIE_DOMAIN` rather than relying on this
-/// function — pulling in the Public Suffix List for one knob isn't
-/// worth it yet.
-pub fn registrable_domain(url: &str) -> Option<String> {
-    let after_scheme = url.split_once("://")?.1;
-    let host_with_port = after_scheme.split('/').next()?;
-    let host = host_with_port
-        .rsplit_once(':')
-        .map_or(host_with_port, |(h, _)| h)
-        .to_ascii_lowercase();
-    if host.is_empty() {
-        return None;
-    }
-    let labels: Vec<&str> = host.split('.').filter(|l| !l.is_empty()).collect();
-    if labels.len() < 2 {
-        // Bare hostname (e.g. `localhost`) — return as-is, no leading
-        // dot. Setting `.localhost` as cookie domain is invalid.
-        return Some(host);
-    }
-    let registrable = &labels[labels.len() - 2..];
-    Some(format!(".{}", registrable.join(".")))
-}
+/// Re-export so existing callers keep working after the helper moved
+/// to `crawler::url_utils`. The body lives there.
+pub use crate::crawler::url_utils::registrable_domain;

 /// Inject the PHPSESSID cookie into the browser's cookie store for the
 /// catalog domain. Must be called before any navigation that depends on
@@ -192,44 +165,8 @@ async fn fetch_probe_html(browser: &Browser, probe_url: &str) -> anyhow::Result<
 mod tests {
    use super::*;

-    #[test]
-    fn registrable_domain_strips_subdomain() {
-        assert_eq!(
-            registrable_domain("https://www.target-site.com/manga/foo/").as_deref(),
-            Some(".target-site.com")
-        );
-        assert_eq!(
-            registrable_domain("https://m.example.org").as_deref(),
-            Some(".example.org")
-        );
-    }
-
-    #[test]
-    fn registrable_domain_keeps_two_label_host() {
-        assert_eq!(
-            registrable_domain("https://example.com/").as_deref(),
-            Some(".example.com")
-        );
-    }
-
-    #[test]
-    fn registrable_domain_handles_port() {
-        assert_eq!(
-            registrable_domain("http://www.foo.bar:8080/x").as_deref(),
-            Some(".foo.bar")
-        );
-    }
-
-    #[test]
-    fn registrable_domain_bare_hostname_no_leading_dot() {
-        // .localhost would be invalid as a cookie Domain.
-        assert_eq!(registrable_domain("http://localhost:5173").as_deref(), Some("localhost"));
-    }
-
-    #[test]
-    fn registrable_domain_returns_none_for_garbage() {
-        assert!(registrable_domain("not a url").is_none());
-    }
+    // registrable_domain tests live in crawler::url_utils now —
+    // it's the canonical home for that helper.

    #[test]
    fn classify_probe_ok_when_logo_and_avatar_present() {