Files
Mangalord/backend/Cargo.toml
MechaCat02 cb34eeb82e feat(analysis): add in-process ocrs OCR backend
Add a CPU-only, English-first OCR engine (the `ocrs` crate on the `rten`
runtime) as an alternative to the local-LLM "vision" backend, which ran
too slowly on the target Raspberry Pi 5.

- `ANALYSIS_BACKEND` (default `ocr`) selects the engine; `vision` keeps the
  existing LLM path. Deploy-time, env-only — not admin-tunable.
- `analysis::ocr` adds an `OcrEngine` trait (testable seam), the `OcrsEngine`
  impl (models loaded once, inference on the blocking pool), and an
  `OcrAnalyzeDispatcher` that plugs into the existing `AnalyzeDispatcher`
  seam and reuses `persist_analysis`, so OCR text lands in `page_ocr_text`
  and the `search_doc` tsvector exactly as the vision path produces them.
- The backend image bakes the two `.rten` models into /models, where
  `OCRS_DETECTION_MODEL` / `OCRS_RECOGNITION_MODEL` default.

Because `/v1/me/page-search` ranks on `search_doc`, OCR text search works
the moment pages are processed. Auto-tagging, scene description and NSFW
flags remain the vision backend's job (deferred).

Bumps to 0.90.0 (minor) in both manifests.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-26 07:12:01 +02:00

75 lines
2.4 KiB
TOML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
[package]
name = "mangalord"
version = "0.90.0"
edition = "2021"
default-run = "mangalord"
[lib]
path = "src/lib.rs"
[[bin]]
name = "mangalord"
path = "src/main.rs"
[[bin]]
name = "crawler"
path = "src/bin/crawler.rs"
[dependencies]
axum = { version = "0.7", features = ["macros", "multipart"] }
tokio = { version = "1", features = ["full"] }
sqlx = { version = "0.8", features = ["runtime-tokio", "postgres", "uuid", "chrono", "macros", "migrate"] }
serde = { version = "1", features = ["derive"] }
serde_json = "1"
uuid = { version = "1", features = ["v4", "serde"] }
chrono = { version = "0.4", features = ["serde"] }
chrono-tz = "0.9"
tracing = "0.1"
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
tower = { version = "0.5", features = ["util"] }
tower-http = { version = "0.6", features = ["trace", "cors"] }
thiserror = "1"
anyhow = "1"
async-trait = "0.1"
dotenvy = "0.15"
argon2 = "0.5"
rand = "0.8"
sha2 = "0.10"
subtle = "2"
base64 = "0.22"
# Image decode + downscale for the analysis worker (keep the page image
# under the local vision model's token budget). Only the manga page formats.
image = { version = "0.25", default-features = false, features = ["jpeg", "png", "webp"] }
axum-extra = { version = "0.9", features = ["cookie", "typed-header"] }
time = "0.3"
infer = "0.16"
tokio-util = { version = "0.7", features = ["io"] }
futures-core = "0.3"
futures-util = "0.3"
bytes = "1"
chromiumoxide = { version = "0.7", features = ["tokio-runtime", "_fetcher-rusttls-tokio"], default-features = false }
sysinfo = { version = "0.32", default-features = false, features = ["system", "component"] }
nix = { version = "0.29", features = ["fs"] }
scraper = "0.20"
reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "socks", "cookies", "stream", "json"] }
ocrs = "0.12"
rten = "0.24"
[dev-dependencies]
tempfile = "3"
tower = { version = "0.5", features = ["util"] }
http-body-util = "0.1"
mime = "0.3"
futures-util = "0.3"
tokio = { version = "1", features = ["test-util"] }
# Trim debug builds: keep line numbers in panics / backtraces but drop the
# full DWARF info (variable-level inspection in gdb/lldb). With a sqlx +
# axum + tokio dep tree the default ("full") leaves backend/target on the
# order of tens of GiB; this typically cuts ~5070% off that.
[profile.dev]
debug = "line-tables-only"
[profile.test]
debug = "line-tables-only"