diff --git a/backend/Cargo.lock b/backend/Cargo.lock index f59fcb5..9342d3d 100644 --- a/backend/Cargo.lock +++ b/backend/Cargo.lock @@ -1517,7 +1517,7 @@ checksum = "c41e0c4fef86961ac6d6f8a82609f55f31b05e4fce149ac5710e439df7619ba4" [[package]] name = "mangalord" -version = "0.79.0" +version = "0.79.1" dependencies = [ "anyhow", "argon2", diff --git a/backend/Cargo.toml b/backend/Cargo.toml index 125c279..8b1bca4 100644 --- a/backend/Cargo.toml +++ b/backend/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "mangalord" -version = "0.79.0" +version = "0.79.1" edition = "2021" default-run = "mangalord" diff --git a/backend/src/analysis/prompt.rs b/backend/src/analysis/prompt.rs index 8fdbdbb..5e0941e 100644 --- a/backend/src/analysis/prompt.rs +++ b/backend/src/analysis/prompt.rs @@ -29,6 +29,13 @@ pub const MAX_OCR_PIECES: usize = 200; pub const MAX_OCR_TEXT_CHARS: usize = 500; pub const MAX_SCENE_CHARS: usize = 1000; +/// Hard per-call caps baked into the JSON schemas (`maxItems`/`maxLength`), +/// so a repetition-prone model is grammar-bounded and can't loop a single +/// call into the token ceiling. Kept per-call/per-slice; the cross-slice +/// total is bounded separately by [`MAX_OCR_PIECES`]. +pub const MAX_OCR_PER_CALL: usize = 50; +pub const MAX_TAGS_PER_CALL: usize = 12; + /// The system prompt. Self-contained: it carries the exact JSON schema so /// the same string drives both the model and (implicitly) the /// [`crate::domain::page_analysis::VisionAnalysis`] parser. @@ -43,13 +50,15 @@ pub const SYSTEM_PROMPT: &str = concat!( "\"safety_flag\":{\"is_nsfw\":boolean,\"content_type\":[", "\"sexual|nudity|gore|violence|disturbing\"]}}\n", "Rules: transcribe every visible text element verbatim into ocr_results ", - "with its kind; if there is no text use []. tagging_results: 5-10 short, ", - "lowercase content tags (characters, actions, setting, mood, genre); ", - "include explicit/sexual tags when present. scene_description: one or two ", - "sentences describing setting, characters, and action. safety_flag.", - "content_type: only values from the listed set, [] if none; set is_nsfw ", - "true if the page contains sexual, nudity, gore, violence, or disturbing ", - "content. Output must be valid minified JSON and nothing else." + "with its kind, each ONCE (never repeat or loop), at most 50; if there is ", + "no text use []. tagging_results: 5-10 short, lowercase, distinct content ", + "tags (characters, actions, setting, mood, genre); include explicit/", + "sexual tags when present. scene_description: one or two sentences ", + "describing setting, characters, and action. safety_flag.content_type: ", + "only values from the listed set, [] if none; set is_nsfw true if the ", + "page contains sexual, nudity, gore, violence, or disturbing content. ", + "CRITICAL: do not repeat elements, tags or words; STOP when done. Output ", + "must be valid minified JSON and nothing else." ); // --- Two-pass prompts (long-page slicing) ----------------------------------- @@ -62,9 +71,11 @@ pub const OCR_PROMPT: &str = concat!( "into ocr_results, each with its kind ", "(speech|thought|narration|sfx|title|caption) and y — the vertical center ", "of the text as a fraction from 0.0 (top) to 1.0 (bottom) of THIS image. ", - "List them in top-to-bottom order. If there is no text, return ", - "{\"ocr_results\":[]}. Output ONLY one minified JSON object, no prose, no ", - "markdown." + "List them ONCE each, in top-to-bottom order. CRITICAL: transcribe each ", + "distinct text element exactly once — never repeat an element, never loop ", + "or pad the output. At most 50 elements. The moment you have transcribed ", + "all visible text, STOP. If there is no text, return {\"ocr_results\":[]}. ", + "Output ONLY one minified JSON object, no prose, no markdown." ); /// Pass B: tags + scene + safety for the whole page, grounded in the merged @@ -74,11 +85,13 @@ pub const GROUNDING_PROMPT: &str = concat!( "(possibly downscaled) AND the OCR text already extracted from it. Using ", "both, return ONLY one minified JSON object with: tagging_results (5-10 ", "short lowercase content tags — characters, actions, setting, mood, ", - "genre; include explicit/sexual tags when present); scene_description ", - "(one or two sentences on setting, characters, and action, referencing ", - "the dialogue where relevant); safety_flag (is_nsfw boolean + ", - "content_type from sexual|nudity|gore|violence|disturbing, [] if none). ", - "No prose, no markdown." + "genre; include explicit/sexual tags when present; each tag distinct, no ", + "duplicates); scene_description (one or two sentences on setting, ", + "characters, and action, referencing the dialogue where relevant); ", + "safety_flag (is_nsfw boolean + content_type from ", + "sexual|nudity|gore|violence|disturbing, [] if none). CRITICAL: do not ", + "repeat tags, words or sentences; keep each field concise and STOP when ", + "done. No prose, no markdown." ); /// Pass-A schema: `{ ocr_results }` only. @@ -89,11 +102,12 @@ pub fn ocr_json_schema() -> serde_json::Value { "properties": { "ocr_results": { "type": "array", + "maxItems": MAX_OCR_PER_CALL, "items": { "type": "object", "additionalProperties": false, "properties": { - "text": { "type": "string" }, + "text": { "type": "string", "maxLength": MAX_OCR_TEXT_CHARS }, "kind": { "type": "string", "enum": OCR_KINDS }, "y": { "type": "number", "minimum": 0, "maximum": 1 } }, @@ -111,8 +125,12 @@ pub fn grounding_json_schema() -> serde_json::Value { "type": "object", "additionalProperties": false, "properties": { - "tagging_results": { "type": "array", "items": { "type": "string" } }, - "scene_description": { "type": "string" }, + "tagging_results": { + "type": "array", + "maxItems": MAX_TAGS_PER_CALL, + "items": { "type": "string", "maxLength": 64 } + }, + "scene_description": { "type": "string", "maxLength": MAX_SCENE_CHARS }, "safety_flag": { "type": "object", "additionalProperties": false, @@ -120,6 +138,7 @@ pub fn grounding_json_schema() -> serde_json::Value { "is_nsfw": { "type": "boolean" }, "content_type": { "type": "array", + "maxItems": CONTENT_WARNINGS.len(), "items": { "type": "string", "enum": CONTENT_WARNINGS } } }, @@ -142,18 +161,23 @@ pub fn output_json_schema() -> serde_json::Value { "properties": { "ocr_results": { "type": "array", + "maxItems": MAX_OCR_PER_CALL, "items": { "type": "object", "additionalProperties": false, "properties": { - "text": { "type": "string" }, + "text": { "type": "string", "maxLength": MAX_OCR_TEXT_CHARS }, "kind": { "type": "string", "enum": OCR_KINDS } }, "required": ["text", "kind"] } }, - "tagging_results": { "type": "array", "items": { "type": "string" } }, - "scene_description": { "type": "string" }, + "tagging_results": { + "type": "array", + "maxItems": MAX_TAGS_PER_CALL, + "items": { "type": "string", "maxLength": 64 } + }, + "scene_description": { "type": "string", "maxLength": MAX_SCENE_CHARS }, "safety_flag": { "type": "object", "additionalProperties": false, @@ -161,6 +185,7 @@ pub fn output_json_schema() -> serde_json::Value { "is_nsfw": { "type": "boolean" }, "content_type": { "type": "array", + "maxItems": CONTENT_WARNINGS.len(), "items": { "type": "string", "enum": CONTENT_WARNINGS } } }, diff --git a/backend/src/analysis/vision.rs b/backend/src/analysis/vision.rs index d97ed1d..f2dbc32 100644 --- a/backend/src/analysis/vision.rs +++ b/backend/src/analysis/vision.rs @@ -36,6 +36,7 @@ pub struct VisionClient { api_key: Option, max_tokens: u32, response_format: ResponseFormat, + frequency_penalty: f64, slice: SliceParams, } @@ -58,6 +59,7 @@ impl VisionClient { api_key: cfg.api_key.clone(), max_tokens: cfg.max_tokens, response_format: cfg.response_format, + frequency_penalty: cfg.frequency_penalty, slice: SliceParams { max_pixels: cfg.max_pixels, min_slice_height: cfg.min_slice_height, @@ -75,8 +77,13 @@ impl VisionClient { // and let the server cope (preserves the prior behavior). let Some(img) = image::load_from_memory(image).ok() else { let url = format!("data:{mime};base64,{}", b64(image)); - let body = - build_request_body(&self.model, self.max_tokens, &url, self.response_format); + let body = build_request_body( + &self.model, + self.max_tokens, + &url, + self.response_format, + self.frequency_penalty, + ); return parse_chat_completion(&self.post_chat(body).await?); }; @@ -89,6 +96,7 @@ impl VisionClient { self.max_tokens, &data_url(&jpeg), self.response_format, + self.frequency_penalty, ); parse_chat_completion(&self.post_chat(body).await?) } @@ -115,6 +123,7 @@ impl VisionClient { &self.model, self.max_tokens, self.response_format, + self.frequency_penalty, &data_url(&jpeg), ); let parsed = parse_chat_completion(&self.post_chat(body).await?)?; @@ -138,6 +147,7 @@ impl VisionClient { &self.model, self.max_tokens, self.response_format, + self.frequency_penalty, &data_url(&whole), &ocr_text, ); @@ -446,11 +456,13 @@ pub fn build_request_body( max_tokens: u32, data_url: &str, response_format: ResponseFormat, + frequency_penalty: f64, ) -> serde_json::Value { build_body( model, max_tokens, response_format, + frequency_penalty, prompt::SYSTEM_PROMPT, "page_analysis", prompt::output_json_schema(), @@ -464,12 +476,14 @@ fn build_ocr_body( model: &str, max_tokens: u32, response_format: ResponseFormat, + frequency_penalty: f64, data_url: &str, ) -> serde_json::Value { build_body( model, max_tokens, response_format, + frequency_penalty, prompt::OCR_PROMPT, "page_ocr", prompt::ocr_json_schema(), @@ -483,6 +497,7 @@ fn build_grounding_body( model: &str, max_tokens: u32, response_format: ResponseFormat, + frequency_penalty: f64, data_url: &str, ocr_text: &str, ) -> serde_json::Value { @@ -491,6 +506,7 @@ fn build_grounding_body( model, max_tokens, response_format, + frequency_penalty, prompt::GROUNDING_PROMPT, "page_grounding", prompt::grounding_json_schema(), @@ -500,11 +516,14 @@ fn build_grounding_body( } /// Build a chat/completions body with a system prompt, an image part, an -/// optional trailing text part, and the requested `response_format`. +/// optional trailing text part, and the requested `response_format`. A +/// non-zero `frequency_penalty` is included to suppress repetition loops. +#[allow(clippy::too_many_arguments)] fn build_body( model: &str, max_tokens: u32, response_format: ResponseFormat, + frequency_penalty: f64, system_prompt: &str, schema_name: &str, schema: serde_json::Value, @@ -525,6 +544,9 @@ fn build_body( { "role": "user", "content": user_content } ] }); + if frequency_penalty.abs() > f64::EPSILON { + body["frequency_penalty"] = json!(frequency_penalty); + } let rf = match response_format { ResponseFormat::None => None, ResponseFormat::JsonObject => Some(json!({ "type": "json_object" })), @@ -574,6 +596,22 @@ fn extract_json_object(s: &str) -> &str { /// content-warning values. pub fn sanitize(mut a: VisionAnalysis) -> VisionAnalysis { a.ocr_results.retain(|r| !r.text.trim().is_empty()); + + // Collapse runaway repetition: a degenerating model emits the same line + // over and over. Drop any piece identical (normalized text + kind) to the + // one just kept, so a loop that slips past the schema caps still can't + // flood the row. + let mut prev: Option = None; + a.ocr_results.retain(|r| { + let key = format!("{}\u{0}{}", r.kind, normalize_text(&r.text)); + if prev.as_deref() == Some(key.as_str()) { + false + } else { + prev = Some(key); + true + } + }); + a.ocr_results.truncate(MAX_OCR_PIECES); for r in &mut a.ocr_results { r.text = truncate_chars(r.text.trim(), MAX_OCR_TEXT_CHARS); @@ -702,7 +740,8 @@ mod tests { #[test] fn combined_body_uses_full_schema_and_image() { - let body = build_request_body("m", 100, "data:image/png;base64,AA", ResponseFormat::JsonSchema); + let body = + build_request_body("m", 100, "data:image/png;base64,AA", ResponseFormat::JsonSchema, 0.3); assert_eq!(body["messages"][0]["role"], "system"); assert_eq!( body["messages"][1]["content"][0]["image_url"]["url"], @@ -714,18 +753,24 @@ mod tests { } #[test] - fn ocr_body_carries_only_the_ocr_schema() { - let body = build_ocr_body("m", 100, ResponseFormat::JsonSchema, "d"); + fn ocr_body_carries_only_the_ocr_schema_and_caps() { + let body = build_ocr_body("m", 100, ResponseFormat::JsonSchema, 0.3, "d"); + let schema = &body["response_format"]["json_schema"]["schema"]; assert_eq!(body["response_format"]["json_schema"]["name"], "page_ocr"); - let req = &body["response_format"]["json_schema"]["schema"]["required"]; - assert_eq!(req, &json!(["ocr_results"])); - // No user text part — just the image. + assert_eq!(schema["required"], json!(["ocr_results"])); + // Hard repetition guards baked into the schema. + assert!(schema["properties"]["ocr_results"]["maxItems"].is_number()); + assert!( + schema["properties"]["ocr_results"]["items"]["properties"]["text"]["maxLength"] + .is_number() + ); assert_eq!(body["messages"][1]["content"].as_array().unwrap().len(), 1); } #[test] fn grounding_body_includes_ocr_context_and_grounding_schema() { - let body = build_grounding_body("m", 100, ResponseFormat::JsonSchema, "d", "[speech] Hi"); + let body = + build_grounding_body("m", 100, ResponseFormat::JsonSchema, 0.3, "d", "[speech] Hi"); assert_eq!(body["response_format"]["json_schema"]["name"], "page_grounding"); let parts = body["messages"][1]["content"].as_array().unwrap(); assert_eq!(parts.len(), 2, "image + text parts"); @@ -736,10 +781,33 @@ mod tests { #[test] fn response_format_none_omits_the_field() { - let body = build_request_body("m", 100, "d", ResponseFormat::None); + let body = build_request_body("m", 100, "d", ResponseFormat::None, 0.3); assert!(body.get("response_format").is_none()); } + #[test] + fn frequency_penalty_included_only_when_nonzero() { + let on = build_ocr_body("m", 100, ResponseFormat::None, 0.4, "d"); + assert_eq!(on["frequency_penalty"], 0.4); + let off = build_ocr_body("m", 100, ResponseFormat::None, 0.0, "d"); + assert!(off.get("frequency_penalty").is_none()); + } + + #[test] + fn sanitize_collapses_runaway_repetition() { + let body = chat_body( + r#"{"ocr_results":[ + {"text":"I don't know","kind":"speech"}, + {"text":"I don't know","kind":"speech"}, + {"text":"I don't know","kind":"speech"}, + {"text":"ok","kind":"speech"} + ]}"#, + ); + let v = parse_chat_completion(&body).unwrap(); + let texts: Vec<&str> = v.ocr_results.iter().map(|o| o.text.as_str()).collect(); + assert_eq!(texts, vec!["I don't know", "ok"], "consecutive repeats collapsed"); + } + // --- geometry: plan_slices --- #[test] diff --git a/backend/src/config.rs b/backend/src/config.rs index f66e89c..30f0368 100644 --- a/backend/src/config.rs +++ b/backend/src/config.rs @@ -141,6 +141,11 @@ pub struct AnalysisConfig { /// Output-constraint mode (`ANALYSIS_RESPONSE_FORMAT`): /// `json_schema` (default) | `json_object` | `none`. pub response_format: ResponseFormat, + /// Sampling `frequency_penalty` sent with each request + /// (`ANALYSIS_FREQUENCY_PENALTY`). A small positive value discourages + /// the repetition loops that otherwise run small models into the token + /// ceiling. `0` omits the field. + pub frequency_penalty: f64, } impl Default for AnalysisConfig { @@ -164,6 +169,7 @@ impl Default for AnalysisConfig { max_slices: 16, max_image_bytes: 8 * 1024 * 1024, response_format: ResponseFormat::JsonSchema, + frequency_penalty: 0.3, } } } @@ -199,6 +205,7 @@ impl AnalysisConfig { response_format: std::env::var("ANALYSIS_RESPONSE_FORMAT") .map(|s| ResponseFormat::from_str(&s)) .unwrap_or(d.response_format), + frequency_penalty: env_f64("ANALYSIS_FREQUENCY_PENALTY", d.frequency_penalty), } } } @@ -619,6 +626,7 @@ mod tests { "ANALYSIS_TALL_ASPECT", "ANALYSIS_MAX_SLICES", "ANALYSIS_RESPONSE_FORMAT", + "ANALYSIS_FREQUENCY_PENALTY", ] { std::env::remove_var(k); } @@ -629,6 +637,7 @@ mod tests { assert_eq!(cfg.min_slice_height, 640); assert_eq!(cfg.max_slices, 16); assert_eq!(cfg.tall_aspect_threshold, 1.6); + assert_eq!(cfg.frequency_penalty, 0.3); assert!(cfg.api_key.is_none()); assert_eq!(cfg.response_format, ResponseFormat::JsonSchema); } diff --git a/frontend/package.json b/frontend/package.json index 4e7f21b..ec8c959 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -1,6 +1,6 @@ { "name": "mangalord-frontend", - "version": "0.79.0", + "version": "0.79.1", "private": true, "type": "module", "scripts": {