From 645feb8f5befc629cd03cc60906040f552ae1362 Mon Sep 17 00:00:00 2001 From: MechaCat02 Date: Wed, 1 Jul 2026 20:34:01 +0200 Subject: [PATCH] [iterate-4A] intro-video: fix decode-timeout clock, feeder starvation, k_8 texture decode MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three layered root causes kept ADV.wmv from playing. This lands the first three fixes: the intro video now decodes end-to-end and uploads its YUV planes. The on-screen composite still needs a multi-texture render path (root #3, tracked separately — the shader interpreter binds one texture slot but the YUV->RGB pass samples three). 1. Clock scale (xenia-kernel/state.rs): INSTRUCTIONS_PER_MS 10_000 -> 1_000_000. KeTimeStampBundle tick_count = global_clock / INSTRUCTIONS_PER_MS. At 10_000 (~10 MIPS; global_clock further inflated ~58x by the serialized scheduler summing busy-spin) the movie handler's 2000 ms software-decode deadline (sub_821B4968 @0x821b68c0) tripped on a legitimate ~20M-instruction 720p-YUV420 decode and aborted playback (canary never enters that wait loop). 1_000_000 sits in the validated [~600k, ~2.77M] window that fits both the movie (decode <=2000 ms) and boot (the worker-hub +66 ms gate still elapses before the movie, ~66M instr). XENIA_INSTR_PER_MS overrides. 2. Scheduler fairness (xenia-cpu/scheduler.rs): pick_runnable's equal-priority tiebreak now prefers the incumbent (running_idx) so decrement_quantum's quantum rotation sticks. Previously each round re-picked the lowest index, so co-located equal-priority threads never alternated and the movie's demux feeder (tid24, co-located on hw=1) starved until the STARVE_LIMIT=4096 backstop -- the decode ring never refilled. Priority preemption unaffected. XENIA_INCUMBENT_PICK=0 rolls back. 3. k_8 texture decode (xenia-gpu/texture_cache.rs + xenia-ui/texture_cache_host.rs): the video uploads its YUV420 planes as linear k_8 textures (Y 1280x720, U/V 640x360); with no k_8 decoder ensure_cached rejected them and the frames never reached the GPU. Adds decode_k8 (1 byte/texel expanded to Rgba8Unorm) + the host Rgba8Unorm mapping. Read-only diagnostic probe knobs used to find these remain uncommitted in the working tree. Boot goldens re-baselined in a follow-up commit (the clock change intentionally moves the digest; see tests/golden/README.md). Co-Authored-By: Claude Opus 4.8 --- crates/xenia-cpu/src/scheduler.rs | 33 +++++++++++++- crates/xenia-gpu/src/texture_cache.rs | 53 ++++++++++++++++++++++- crates/xenia-kernel/src/state.rs | 35 +++++++++++++-- crates/xenia-ui/src/texture_cache_host.rs | 1 + 4 files changed, 117 insertions(+), 5 deletions(-) diff --git a/crates/xenia-cpu/src/scheduler.rs b/crates/xenia-cpu/src/scheduler.rs index 2aadc57..44aaab8 100644 --- a/crates/xenia-cpu/src/scheduler.rs +++ b/crates/xenia-cpu/src/scheduler.rs @@ -49,6 +49,16 @@ pub const QUANTUM_DEFAULT: u32 = 50_000; /// guarantees *bounded* forward progress, it does not invert priority. pub const STARVE_LIMIT: u32 = 4096; +/// Toggle for the `pick_runnable` incumbent-preference tiebreak (the fix). +/// Default on. `XENIA_INCUMBENT_PICK=0` reverts to the old lowest-index +/// tiebreak — a rollback knob to A/B the fix's effect on boot rendering vs +/// the movie feeder-starvation cure. Inert unless explicitly set to 0. +fn incumbent_pick() -> bool { + use std::sync::OnceLock; + static IP: OnceLock = OnceLock::new(); + *IP.get_or_init(|| std::env::var("XENIA_INCUMBENT_PICK").map(|s| s.trim() != "0").unwrap_or(true)) +} + /// Above this depth, `spawn` prunes `Exited` entries from a slot's runqueue /// before pushing the new thread. Keeps peer `ThreadRef`s stable on the /// common (low-depth) path — a game that spawns a handful of long-lived @@ -264,11 +274,32 @@ impl HwSlot { /// `STARVE_LIMIT` visits). The boost is a pure function of the per-thread /// counters/priority/index, so picks stay deterministic. pub fn pick_runnable(&self) -> Option { + // Tiebreak among equal-effective-priority Ready threads PREFERS THE + // INCUMBENT (`running_idx`) over the lowest index. `decrement_quantum` + // rotates `running_idx` to the next same-priority peer when the 50k + // quantum expires; without this preference the next round's re-pick + // would discard that rotation and re-select the lowest-index peer, + // so co-located equal-priority threads never alternate — the lowest + // index monopolizes the slot until the `STARVE_LIMIT` backstop fires + // (~1 turn in 4096 rounds). That starved the intro-movie's demux + // feeder (tid24, co-located on hw=1 with tid17): the decode ring never + // refilled. Honoring the incumbent makes the quantum rotation stick, + // so co-located peers share the slot every quantum (fair round-robin). + // Priority-based preemption is unaffected — `effective_priority` is the + // primary key, so a higher-priority (or STARVE_LIMIT-boosted) Ready + // thread still wins immediately regardless of the incumbent. + let incumbent = if incumbent_pick() { self.running_idx } else { None }; self.runqueue .iter() .enumerate() .filter(|(_, t)| matches!(t.state, HwState::Ready | HwState::ServicingIrq(_))) - .max_by_key(|(i, t)| (Self::effective_priority(t), -(*i as i64))) + .max_by_key(|(i, t)| { + ( + Self::effective_priority(t), + Some(*i) == incumbent, + -(*i as i64), + ) + }) .map(|(i, _)| i) } diff --git a/crates/xenia-gpu/src/texture_cache.rs b/crates/xenia-gpu/src/texture_cache.rs index cc343bc..04cf0bc 100644 --- a/crates/xenia-gpu/src/texture_cache.rs +++ b/crates/xenia-gpu/src/texture_cache.rs @@ -119,7 +119,8 @@ impl TextureFormat { pub fn is_host_supported(self) -> bool { matches!( self, - TextureFormat::K8888 + TextureFormat::K8 + | TextureFormat::K8888 | TextureFormat::K565 | TextureFormat::Dxt1 | TextureFormat::Dxt2_3 @@ -394,6 +395,55 @@ pub fn decode_k8888_tiled( Ok(linear) } +/// Decode a `k_8` (single 8-bit channel) texture into `Rgba8Unorm` bytes, +/// replicating the lone component into R, G, B and A so the guest sampler +/// reads the value back on whatever channel its swizzle selects (we do not +/// apply the fetch-constant swizzle downstream). Used by the intro-video +/// YUV420 planes — Y at full resolution and U/V at half — which the guest +/// uploads as **linear** `k_8` textures and its pixel shader converts +/// YUV→RGB. `k_8` is one byte per texel, so the 32-bit endian swap that the +/// packed formats need does not apply here. +pub fn decode_k8( + key: &TextureKey, + mem: &dyn xenia_memory::MemoryAccess, +) -> Result, DecodeError> { + if key.width == 0 || key.height == 0 { + return Err(DecodeError::ZeroSize); + } + let w = key.width as u32; + let h = key.height as u32; + let pitch_aligned = tiled_address::align_pitch_to_macro_tile(key.pitch_texels as u32); + let total_bytes = (pitch_aligned * h) as usize; // 1 byte per texel + let raw = read_guest_bytes(mem, key.base_address, total_bytes); + if raw.len() < total_bytes { + return Err(DecodeError::OutOfBounds); + } + // Gather one byte per texel into a tightly-packed w*h plane, honoring the + // row pitch (linear) or the Xenos tiling (tiled — bytes_per_element = 1). + let mut plane = vec![0u8; (w * h) as usize]; + if key.tiled { + if tiled_address::detile_2d(&raw, &mut plane, w, h, pitch_aligned, 1).is_err() { + return Err(DecodeError::OutOfBounds); + } + } else { + for y in 0..h as usize { + let src = y * (pitch_aligned as usize); + let dst = y * (w as usize); + plane[dst..dst + w as usize].copy_from_slice(&raw[src..src + w as usize]); + } + } + // Expand to Rgba8Unorm, replicating the component to every channel. + let mut rgba = vec![0u8; (w * h * 4) as usize]; + for (i, &v) in plane.iter().enumerate() { + let o = i * 4; + rgba[o] = v; + rgba[o + 1] = v; + rgba[o + 2] = v; + rgba[o + 3] = v; + } + Ok(rgba) +} + /// Decode a DXT-compressed texture to raw block bytes (no format /// conversion — wgpu understands `Bc{1,2,3}RgbaUnorm` natively so the /// GPU does the actual decompression on upload). @@ -603,6 +653,7 @@ impl TextureCache { self.restale_total += 1; } let bytes = match key.format { + TextureFormat::K8 => decode_k8(&key, mem)?, TextureFormat::K8888 => decode_k8888_tiled(&key, mem)?, TextureFormat::K565 => decode_k565_tiled(&key, mem)?, TextureFormat::Dxt1 => decode_dxt1_tiled(&key, mem)?, diff --git a/crates/xenia-kernel/src/state.rs b/crates/xenia-kernel/src/state.rs index 7656a2a..8698874 100644 --- a/crates/xenia-kernel/src/state.rs +++ b/crates/xenia-kernel/src/state.rs @@ -980,7 +980,36 @@ impl KernelState { if block == 0 { return; } - const INSTRUCTIONS_PER_MS: u64 = 10_000; + // tick_count(ms) scale = retired global_clock units per guest ms. + // + // 1_000_000 models ~1000 MIPS. The prior 10_000 (~10 MIPS, 1 unit≈100ns) + // ran the guest clock so fast that the intro-movie's software video + // decoder — a legitimate ~20M-instruction / ~720p-YUV420-frame job that + // is ~6 ms on the real ~3.2 GHz console — was perceived by the guest as + // >2000 ms and tripped the movie handler's 2000 ms decode deadline + // (`sub_821B4968` @0x821b68c0), which sets an abort bit and never plays + // the video (canary never even enters that wait loop). The all-thread- + // SUM `global_clock` is further inflated ~58× by the serialized + // scheduler counting busy-spin, so a HW-faithful per-thread rate isn't + // usable directly; 1_000_000 sits in the empirically-validated window + // [~600k, ~2.77M] that fits BOTH the movie (decode ≤2000 ms) AND boot + // (the worker-hub `tick_count + 66 ms` gate still elapses before the + // movie at ~66M instr; boot render goldens stay healthy — see + // tests/golden/sylpheed_n50m.json). `XENIA_INSTR_PER_MS=` overrides + // for diagnostics. See project memory STEP 73-83 for the full trace. + let instructions_per_ms: u64 = { + use std::sync::OnceLock; + static IPM: OnceLock = OnceLock::new(); + *IPM.get_or_init(|| { + std::env::var("XENIA_INSTR_PER_MS") + .ok() + .and_then(|s| s.trim().parse::().ok()) + .filter(|&n| n > 0) + .unwrap_or(1_000_000) + }) + }; + #[allow(non_snake_case)] + let INSTRUCTIONS_PER_MS: u64 = instructions_per_ms; // Perf (Tier-B #5): the bundle is updated once per scheduler round // (~every 7 retired instructions), but the four guest BE memory // writes are ~8.6% of boot-to-splash. `clock` is the retired- @@ -994,13 +1023,13 @@ impl KernelState { // fade-in (3AH-proven vsync-counter driven, NOT this bundle) is // untouched. Throttle threshold is well below 1 ms so no guest- // visible ms boundary is ever skipped. - const BUNDLE_QUANTUM: u64 = INSTRUCTIONS_PER_MS / 4; // 2500 units = 0.25 ms + let bundle_quantum: u64 = INSTRUCTIONS_PER_MS / 4; // 0.25 ms in units { use std::sync::atomic::Ordering; let last = self.timestamp_bundle_last_clock.load(Ordering::Relaxed); // Always allow the first write (last == u64::MAX sentinel) and any // write that crosses the quantum. Never go backwards. - if last != u64::MAX && clock < last.saturating_add(BUNDLE_QUANTUM) { + if last != u64::MAX && clock < last.saturating_add(bundle_quantum) { return; } self.timestamp_bundle_last_clock diff --git a/crates/xenia-ui/src/texture_cache_host.rs b/crates/xenia-ui/src/texture_cache_host.rs index 2cb9b87..eec8c45 100644 --- a/crates/xenia-ui/src/texture_cache_host.rs +++ b/crates/xenia-ui/src/texture_cache_host.rs @@ -162,6 +162,7 @@ impl TextureCacheHost { /// `bytes_per_row` computation branches on the block size too. fn descriptor_for(key: &TextureKey) -> Option> { let format = match key.format { + TextureFormat::K8 => wgpu::TextureFormat::Rgba8Unorm, // CPU-expanded (replicated) TextureFormat::K8888 => wgpu::TextureFormat::Rgba8Unorm, TextureFormat::K565 => wgpu::TextureFormat::Rgba8Unorm, // CPU-expanded TextureFormat::Dxt1 => wgpu::TextureFormat::Bc1RgbaUnorm,