Cutscene voice: resolve movies via continuous-stream cue regions

Cracked how the game maps a movie to its voice track and reimplemented it,
fixing wrong/short voices for RT and hokyu cutscenes.

## Mechanism (RE'd from a Canary file-I/O trace, user-confirmed by ear)

The movie voices are ONE continuous XMA stream, physically chunked into
`<lang>\Movie\VOICE_*.slb` (and `\etc\VOICE_D_*.slb`) sound.pak TOC entries
whose boundaries do NOT align with the cutscene cues. A single cue routinely
spans two `.slb` chunks, so a `.slb` does not necessarily hold the track its
name claims (they are off-by-one vs the cues). Each cue ends at an inline
trailer descriptor `(sound_id:u32be, 0x11, …)` whose id repeats at +0x800.

Resolution chain (all derivable on-disc, no guest-code RE):
  1. movie -> cue token         (ADVERTISE_MOVIE manifest VOICETRACK)
  2. cue token -> sound-id       (master sound registry, tables.pak
                                  a2c8c185=eng / b04238e0=jpn, schema 0x13cb84ba)
  3. sound-id -> [start,end)      scan the stream for trailer(id) and its
                                  predecessor; cue audio = the bytes between,
                                  read continuously across chunk boundaries.

Validated: decoded voice length matches each movie within ~0.4s, including
cross-boundary cues (ADV/RT01A/RT01B/RT01C/S00A/S01A + the 5 bound hokyu).

## Changes

- sylpheed-formats/src/movie_voice.rs (new): registry_voice_ids(),
  find_descriptor(), find_descriptor_before() (gap-tolerant predecessor).
  Unit-tested.
- viewer iso_loader.rs: resolve_movie_voice_region() + decode_voice_region()
  (continuous read via existing read_segment_range + to_xma_riffs). Voice/
  audio requests now region-first; a `\Movie\` token that fails region stays
  SILENT (never plays the wrong off-by-one chunk). Removed the old
  movie_is_demo_based suppress hack.
- Anchor tries subdirs Movie/etc/Voice so hokyu `\etc\VOICE_D_*` resolve too.
- Examples resolve_all/validate_cues/hokyu_final document + validate the pipeline.

## Status

- RT / S / ADV cutscene voices: CORRECT (user-confirmed).
- 5 manifest-bound hokyu (VOICE_D_450-454): CORRECT (user-confirmed).

## OPEN (handoff)

13 of 18 hokyu movies have NO manifest VOICETRACK; the game reuses the 5
recordings across stages via a code rule. `hokyu_fallback_token()` currently
guesses by category (ship LS/DS x source carrier-A/tanker-H: LS/A->450,
LS/H->453, DS/A->452, DS/H->454). USER REPORTS THIS IS SOMETIMES WRONG.
- LS/carrier has two takes (450 from s02A, 451 from s09A); the early-vs-late
  split is unknown — unbound LS/A currently all default to 450.
- Definitive fix = Canary `--phase_a_fileio_only` file.read trace of an unbound
  hokyu (e.g. hokyu_LS_s03A) to see which VOICE_D the game actually streams,
  then encode the real rule. Same method that cracked the RT mapping.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
MechaCat02
2026-07-23 06:53:55 +02:00
parent ed5c2f24c3
commit 053aa81953
9 changed files with 577 additions and 54 deletions

View File

@@ -0,0 +1,36 @@
use sylpheed_formats::{hash::name_hash, movie_manifest, movie_voice, slb, PakArchive};
use std::fs;use std::io::{Read,Seek,SeekFrom};use std::process::Command;
fn rg(disc:&str,g:u64,n:usize)->Vec<u8>{let mut segs=vec![];let mut cum=0u64;for i in 0..5{let p=format!("{disc}/dat/sound.p{i:02}");if let Ok(m)=fs::metadata(&p){segs.push((cum,m.len(),p));cum+=m.len();}}let mut out=vec![];let(mut need,mut pos)=(n,g);for(base,len,path)in &segs{if need==0||pos>=base+len||pos<*base{continue;}let local=pos-base;let take=need.min((len-local)as usize);let mut f=fs::File::open(path).unwrap();f.seek(SeekFrom::Start(local)).unwrap();let mut b=vec![0u8;take];f.read_exact(&mut b).unwrap();out.extend_from_slice(&b);need-=take;pos+=take as u64;}out}
fn contains(h:&[u8],n:&[u8])->bool{h.windows(n.len()).any(|w|w==n)}
fn dur(w:&str)->String{let o=Command::new("ffprobe").args(["-v","error","-show_entries","format=duration","-of","default=nw=1:nk=1",w]).output().unwrap();String::from_utf8_lossy(&o.stdout).trim().to_string()}
fn hokyu_fallback(m:&str)->Option<String>{if !m.starts_with("hokyu_"){return None}let ls=m.contains("_LS_");let ds=m.contains("_DS_");let a=m.ends_with('A');let h=m.ends_with('H');Some(match(ls,ds,a,h){(true,_,true,_)=>"VOICE_D_450",(true,_,_,true)=>"VOICE_D_453",(_,true,true,_)=>"VOICE_D_452",(_,true,_,true)=>"VOICE_D_454",_=>return None}.into())}
fn main(){
let disc=std::env::var("SYLPHEED_DISC").unwrap();
let tpak=PakArchive::open(format!("{disc}/dat/tables.pak")).unwrap();
let manifest=tpak.entries().iter().find_map(|e|tpak.read(e).ok().filter(|b|movie_manifest::is_manifest(b))).unwrap();
let code="eng";
let registry=tpak.entries().iter().find_map(|e|tpak.read(e).ok().filter(|b|contains(b,format!("{code}\\Movie\\VOICE_ADV.slb").as_bytes()))).unwrap();
let ids=movie_voice::registry_voice_ids(&registry);
let stoc=fs::read(format!("{disc}/dat/sound.pak")).unwrap();
let entries=PakArchive::parse_toc(&stoc).unwrap();
let hokyu:Vec<String>=movie_manifest::parse(&manifest).into_iter().filter(|m|m.movie.starts_with("hokyu_")).map(|m|m.movie).collect();
for movie in &hokyu{
let bound=movie_manifest::voice_token(&manifest,movie);
let token=bound.clone().or_else(||hokyu_fallback(movie));
let Some(token)=token else{println!("{movie:16} NO TOKEN"); continue};
let src = if bound.is_some(){"bound"}else{"fallback"};
let Some(&id)=ids.get(&token) else{println!("{movie:16} {token} not in registry");continue};
let Some(anchor)=["Movie","etc","Voice"].iter().find_map(|d|{let h=name_hash(&format!("{code}\\{d}\\{token}.slb"));entries.binary_search_by_key(&h,|e|e.name_hash).ok().map(|i|entries[i].offset as u64)}) else{println!("{movie:16} no anchor");continue};
let win_start=(anchor.saturating_sub(2*1024*1024))&!3;
let window=rg(&disc,win_start,8*1024*1024);
let Some(el)=movie_voice::find_descriptor(&window,id) else{println!("{movie:16} desc not found");continue};
let end=win_start+el as u64;
let start=movie_voice::find_descriptor(&window,id.wrapping_sub(1)).or_else(||movie_voice::find_descriptor_before(&window,el)).map(|o|win_start+o as u64).filter(|&s|s<end&&end-s<1_500_000).unwrap_or(anchor);
let region=rg(&disc,start,(end-start)as usize);
let mut riffs=slb::to_xma_riffs(&region); if riffs.is_empty(){riffs=slb::to_xma_riff_best(&region).into_iter().collect();}
let mut d="?".into();
if let Some(r)=riffs.first(){let xp=format!("/tmp/hf_{movie}.xma.wav");let wp=format!("/tmp/hf_{movie}.wav");fs::write(&xp,r).unwrap();let _=Command::new("ffmpeg").args(["-hide_banner","-v","error","-y","-i",&xp,&wp]).status();d=dur(&wp);}
let mv=dur(&format!("{disc}/dat/movie/{movie}.wmv"));
println!("{movie:16} {src:8} {token:12} id={id} voice={d}s movie={mv}s");
}
}

View File

@@ -0,0 +1,36 @@
use sylpheed_formats::{hash::name_hash, movie_manifest, movie_voice, slb, PakArchive};
use std::fs;use std::io::{Read,Seek,SeekFrom};use std::process::Command;
fn rg(disc:&str,g:u64,n:usize)->Vec<u8>{let mut segs=vec![];let mut cum=0u64;for i in 0..5{let p=format!("{disc}/dat/sound.p{i:02}");if let Ok(m)=fs::metadata(&p){segs.push((cum,m.len(),p));cum+=m.len();}}let mut out=vec![];let(mut need,mut pos)=(n,g);for(base,len,path)in &segs{if need==0||pos>=base+len||pos<*base{continue;}let local=pos-base;let take=need.min((len-local)as usize);let mut f=fs::File::open(path).unwrap();f.seek(SeekFrom::Start(local)).unwrap();let mut b=vec![0u8;take];f.read_exact(&mut b).unwrap();out.extend_from_slice(&b);need-=take;pos+=take as u64;}out}
fn contains(h:&[u8],n:&[u8])->bool{h.windows(n.len()).any(|w|w==n)}
fn dur(w:&str)->String{let o=Command::new("ffprobe").args(["-v","error","-show_entries","format=duration","-of","default=nw=1:nk=1",w]).output().unwrap();String::from_utf8_lossy(&o.stdout).trim().to_string()}
fn main(){
let disc=std::env::var("SYLPHEED_DISC").unwrap();
let tpak=PakArchive::open(format!("{disc}/dat/tables.pak")).unwrap();
let manifest=tpak.entries().iter().find_map(|e|tpak.read(e).ok().filter(|b|movie_manifest::is_manifest(b))).unwrap();
let code="eng";
let registry=tpak.entries().iter().find_map(|e|tpak.read(e).ok().filter(|b|contains(b,format!("{code}\\Movie\\VOICE_ADV.slb").as_bytes()))).unwrap();
let ids=movie_voice::registry_voice_ids(&registry);
let stoc=fs::read(format!("{disc}/dat/sound.pak")).unwrap();
let entries=PakArchive::parse_toc(&stoc).unwrap();
for movie in ["ADV","RT01A","RT01B","RT01C_1","S00A","S01A","hokyu_LS_s02A","hokyu_LS_s09A","hokyu_LS_s02H","hokyu_DS_s07H"]{
let Some(token)=movie_manifest::voice_token(&manifest,movie) else { println!("{movie:16} no token"); continue };
let Some(&id)=ids.get(&token) else { println!("{movie:16} token {token} not in registry"); continue };
let Some(anchor)=["Movie","etc","Voice"].iter().find_map(|d|{let h=name_hash(&format!("{code}\\{d}\\{token}.slb"));entries.binary_search_by_key(&h,|e|e.name_hash).ok().map(|i|entries[i].offset as u64)}) else { println!("{movie:16} no anchor for {token}"); continue };
let win_start=(anchor.saturating_sub(2*1024*1024))&!3;
let window=rg(&disc,win_start,8*1024*1024);
let Some(end_local)=movie_voice::find_descriptor(&window,id) else { println!("{movie:16} id {id}: desc NOT FOUND"); continue };
let end=win_start+end_local as u64;
let start=movie_voice::find_descriptor(&window,id.wrapping_sub(1))
.or_else(||movie_voice::find_descriptor_before(&window,end_local))
.map(|o|win_start+o as u64)
.filter(|&s|s<end && end-s<1_500_000)
.unwrap_or(anchor);
let region=rg(&disc,start,(end-start) as usize);
let mut riffs=slb::to_xma_riffs(&region);
if riffs.is_empty(){ riffs=slb::to_xma_riff_best(&region).into_iter().collect(); }
let mut d="?".into();
if let Some(r)=riffs.first(){ let xp=format!("/tmp/ra2_{movie}.xma.wav");let wp=format!("/tmp/ra2_{movie}.wav");fs::write(&xp,r).unwrap();let _=Command::new("ffmpeg").args(["-hide_banner","-v","error","-y","-i",&xp,&wp]).status();d=dur(&wp);}
let mv=dur(&format!("{disc}/dat/movie/{movie}.wmv"));
println!("{movie:16} id={id:5} region={}KB riffs={} voice={d}s movie={mv}s", (end-start)/1024, riffs.len());
}
}

View File

@@ -0,0 +1,32 @@
use sylpheed_formats::slb;
use std::fs;use std::io::{Read,Seek,SeekFrom};use std::process::Command;
fn rg(disc:&str,g:u64,n:usize)->Vec<u8>{let mut segs=vec![];let mut cum=0u64;for i in 0..5{let p=format!("{disc}/dat/sound.p{i:02}");if let Ok(m)=fs::metadata(&p){segs.push((cum,m.len(),p));cum+=m.len();}}let mut out=vec![];let(mut need,mut pos)=(n,g);for(base,len,path)in &segs{if need==0||pos>=base+len||pos<*base{continue;}let local=pos-base;let take=need.min((len-local)as usize);let mut f=fs::File::open(path).unwrap();f.seek(SeekFrom::Start(local)).unwrap();let mut b=vec![0u8;take];f.read_exact(&mut b).unwrap();out.extend_from_slice(&b);need-=take;pos+=take as u64;}out}
fn dur(w:&str)->String{let o=Command::new("ffprobe").args(["-v","error","-show_entries","format=duration:stream=channels","-of","default=nw=1:nk=1",w]).output().unwrap();String::from_utf8_lossy(&o.stdout).replace('\n'," ")}
fn movdur(disc:&str,m:&str)->String{ dur(&format!("{disc}/dat/movie/{m}")) }
fn main(){
let disc=std::env::var("SYLPHEED_DISC").unwrap();
// descriptor trailer global offsets (from scan) => cue N data = [desc(N-1)..desc(N)]
// (id, end_off, movie)
let cues=[(1600u32,437044592u64,"ADV.wmv"),(1601,437345648,"RT01A.wmv"),(1602,437712240,"RT01B.wmv"),
(1603,438080880,"RT01C_1.wmv"),(1604,438451568,"RT01C_2.wmv"),(1605,438789488,"RT02A.wmv")];
for k in 1..cues.len(){
let (id,end,mov)=cues[k];
let start=cues[k-1].1;
let region=rg(&disc,start,(end-start) as usize + 12000); // +tail to include full data before next trailer
let riffs=slb::to_xma_riffs(&region);
let mut total=0.0f32; let mut parts=vec![];
for (j,r) in riffs.iter().enumerate(){
let xp=format!("/tmp/cue{id}_{j}.xma.wav"); let wp=format!("/tmp/cue{id}_{j}.wav");
fs::write(&xp,r).unwrap();
let _=Command::new("ffmpeg").args(["-hide_banner","-v","error","-y","-i",&xp,&wp]).status();
let d=dur(&wp); parts.push(d.clone());
total+=d.split_whitespace().next().unwrap_or("0").parse::<f32>().unwrap_or(0.0);
}
// spanning?
let span_adv_end=437547264u64;
let spans = start < span_adv_end && end > span_adv_end || (start/1_000 != end/1_000);
println!("cue {id} ({mov}): region[{start}..{end}] {} bytes, {} riff(s), Σ={:.1}s | movie={} parts={:?}",
end-start, riffs.len(), total, movdur(&disc,mov), parts);
}
println!("\nWAVs at /tmp/cue16XX_*.wav (listen: RT01B=1602, RT01C_1=1603, RT01C_2=1604)");
}

View File

@@ -59,6 +59,8 @@ pub mod slb;
// The movie manifest: authoritative movie → subtitle → voice index. // The movie manifest: authoritative movie → subtitle → voice index.
pub mod movie_manifest; pub mod movie_manifest;
pub mod movie_voice;
/// Re-export the most commonly used types at the crate root. /// Re-export the most commonly used types at the crate root.
pub use font::FontInfo; pub use font::FontInfo;
pub use idxd::{IdxdError, IdxdObject}; pub use idxd::{IdxdError, IdxdObject};

View File

@@ -162,6 +162,43 @@ pub fn track_demo_ids(lang_pak: &PakArchive, movie_basename: &str) -> Vec<u32> {
.collect() .collect()
} }
/// Per-line `(demo id, start seconds)` for a movie's timing track — the radio-box
/// voice schedule. Each `MSG_DEMO_<id>` reference takes the start time of the
/// timing that follows it (a caption may list two demo lines under one timing;
/// both then share that start). Empty for inline-text / absent tracks (those are
/// story movies whose voice is the single manifest `VOICETRACK`, not per-line).
pub fn track_voice_cues(lang_pak: &PakArchive, movie_basename: &str) -> Vec<(u32, f32)> {
let Some(Ok(track)) = lang_pak.read_by_hash(track_key(movie_basename)) else {
return Vec::new();
};
if !is_ixud(&track) {
return Vec::new();
}
let toks = utf16le_tokens(&track);
let start = toks
.iter()
.position(|t| t.ends_with("SUBTITLE"))
.map(|i| i + 1)
.unwrap_or(0);
let mut out = Vec::new();
let mut pending: Vec<u32> = Vec::new();
for tok in &toks[start..] {
match parse_timing(tok) {
Some((s, _)) => {
for d in pending.drain(..) {
out.push((d, s));
}
}
None => {
if let Some(d) = demo_ref(tok) {
pending.push(d);
}
}
}
}
out
}
/// Build `demo id → ordered caption lines` from a `GP_MAIN_GAME_<L>` pack. /// Build `demo id → ordered caption lines` from a `GP_MAIN_GAME_<L>` pack.
/// ///
/// Scans every IXUD block, pairing each text token with the immediately /// Scans every IXUD block, pairing each text token with the immediately

View File

@@ -0,0 +1,142 @@
//! Cutscene-voice resolution: movie → sound-id → continuous-stream byte region.
//!
//! Reverse-engineered 2026-07-22 (Canary file-I/O trace, user-confirmed by ear).
//! The movie voices form **one continuous XMA stream**, physically chunked into
//! `<lang>\Movie\VOICE_*.slb` `sound.pak` TOC entries whose boundaries do **not**
//! align with the cutscene cues — a single cue routinely spans two `.slb` chunks,
//! so a `.slb` does not necessarily hold the track its name claims. Each cue's
//! audio ends at an inline **trailer descriptor** `(sound_id: u32be, 0x11, …)`
//! whose id repeats at +0x800. Cue N's audio = the stream bytes
//! `[descriptor(N-1) .. descriptor(N)]`, read continuously across chunk
//! boundaries. Verified: decoded voice length matches each movie within ~0.3 s,
//! including cross-boundary cues (RT01B/RT01C span `VOICE_ADV`→`VOICE_RT01A`→…).
//!
//! Resolution chain:
//! 1. movie → cue name (`ADVERTISE_MOVIE` manifest `VOICETRACK`, e.g. `VOICE_RT01A`)
//! 2. cue name → sound-id (master sound registry, [`registry_voice_ids`], → 1601)
//! 3. sound-id → `[start, end]` global offsets via [`find_descriptor`] over the
//! `sound.pNN` stream (the caller supplies the bytes + the physical anchor).
use std::collections::HashMap;
/// Parse the master sound registry — the large per-language `tables.pak` IDXD
/// entry (schema `0x13cb84ba`) that carries the `<lang>\Movie\VOICE_*.slb` paths —
/// into `cue-name → sound-id` from its `NAME\0 ID\0` string pool
/// (e.g. `VOICE_ADV`→1600, `VOICE_RT01A`→1601).
pub fn registry_voice_ids(registry: &[u8]) -> HashMap<String, u32> {
let mut toks: Vec<&[u8]> = Vec::new();
let mut start = 0usize;
for i in 0..registry.len() {
if registry[i] == 0 {
if i > start {
toks.push(&registry[start..i]);
}
start = i + 1;
}
}
let mut out = HashMap::new();
for w in toks.windows(2) {
let (name, num) = (w[0], w[1]);
if name.starts_with(b"VOICE_")
&& !num.is_empty()
&& num.iter().all(u8::is_ascii_digit)
{
if let (Ok(n), Ok(id)) = (
std::str::from_utf8(name),
std::str::from_utf8(num).unwrap().parse::<u32>(),
) {
out.entry(n.to_string()).or_insert(id);
}
}
}
out
}
/// The constant marker following a cue's sound-id in its inline trailer.
const DESC_MARK: u32 = 0x11;
/// The trailer repeats its id this many bytes later; used to reject a coincidental
/// `(id, 0x11)` pair inside XMA audio (probability of a false match is ~2⁻⁶⁴).
const DESC_REPEAT: usize = 0x800;
/// Byte offset (within `buf`, a slice of the continuous voice stream) of cue
/// `id`'s trailer descriptor: the first big-endian `(id, 0x11)` pair whose id
/// repeats at `+0x800`. `None` if absent in the slice.
pub fn find_descriptor(buf: &[u8], id: u32) -> Option<usize> {
if buf.len() < DESC_REPEAT + 4 {
return None;
}
let be = |o: usize| u32::from_be_bytes([buf[o], buf[o + 1], buf[o + 2], buf[o + 3]]);
let end = buf.len() - (DESC_REPEAT + 4);
let mut o = 0;
while o <= end {
if be(o) == id && be(o + 4) == DESC_MARK && be(o + DESC_REPEAT) == id {
return Some(o);
}
o += 4;
}
None
}
/// Plausible sound-id range for a trailer (all real cue ids are well under this).
const ID_MAX: u32 = 0x1_0000;
/// Offset of the nearest cue trailer strictly **before** `before` — any plausible
/// id. Used to bound a cue's START when its `id-1` predecessor doesn't exist
/// (the sound-id sequence has gaps, e.g. `VOICE_D_453`=7142 → `VOICE_D_454`=7188).
/// Scans downward and returns the first (largest-offset) valid `(id, 0x11)` with
/// the `+0x800` id-repeat. `None` if there is no trailer below `before`.
pub fn find_descriptor_before(buf: &[u8], before: usize) -> Option<usize> {
if before < 4 {
return None;
}
let cap = buf.len().checked_sub(DESC_REPEAT + 4)?;
let be = |o: usize| u32::from_be_bytes([buf[o], buf[o + 1], buf[o + 2], buf[o + 3]]);
// Start strictly below `before` (4-aligned) so the target's own trailer is
// never returned as its predecessor.
let mut o = (before - 1).min(cap) & !3;
loop {
let id = be(o);
if id >= 1 && id < ID_MAX && be(o + 4) == DESC_MARK && be(o + DESC_REPEAT) == id {
return Some(o);
}
if o < 4 {
return None;
}
o -= 4;
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn registry_parses_name_id_pairs() {
let mut b = Vec::new();
for tok in [
b"VOICE_ADV".as_slice(),
b"1600",
b"VOICE_RT01A",
b"1601",
b"NOT_A_VOICE",
b"9",
] {
b.extend_from_slice(tok);
b.push(0);
}
let ids = registry_voice_ids(&b);
assert_eq!(ids.get("VOICE_ADV"), Some(&1600));
assert_eq!(ids.get("VOICE_RT01A"), Some(&1601));
assert_eq!(ids.get("NOT_A_VOICE"), None);
}
#[test]
fn find_descriptor_matches_repeat() {
let mut b = vec![0u8; 0x900];
b[0..4].copy_from_slice(&1601u32.to_be_bytes());
b[4..8].copy_from_slice(&0x11u32.to_be_bytes());
b[DESC_REPEAT..DESC_REPEAT + 4].copy_from_slice(&1601u32.to_be_bytes());
assert_eq!(find_descriptor(&b, 1601), Some(0));
assert_eq!(find_descriptor(&b, 1600), None);
}
}

View File

@@ -200,6 +200,38 @@ pub fn to_xma_riff(slb: &[u8]) -> Option<Vec<u8>> {
} }
} }
/// Rebuild a standalone XMA1 `RIFF/WAVE` from a single-stream `.slb`, robust to
/// the layout variants seen in `<lang>\etc\` radio clips. Unlike [`to_xma_riff`]
/// (which assumes `RIFF → fmt → data` in order), this picks the **largest `data`
/// chunk anywhere** in the bank — some radio banks store the audio *before* the
/// trailing `RIFF`/`fmt` metadata (an empty post-`RIFF` `data` chunk), which the
/// ordered scan misses. Pairs it with the first `fmt ` chunk; falls back to the
/// headerless layout. Returns `None` only when no usable audio can be found.
pub fn to_xma_riff_best(slb: &[u8]) -> Option<Vec<u8>> {
// Largest usable `data` chunk (bounded by its declared size and the buffer).
let mut best: Option<(usize, usize)> = None; // (data offset, usable payload len)
let mut i = 0;
while let Some(di) = find(slb, b"data", i) {
let declared = le32(slb, di + 4).unwrap_or(0) as usize;
let usable = declared.min(slb.len().saturating_sub(di + 8));
if best.map_or(true, |(_, b)| usable > b) {
best = Some((di, usable));
}
i = di + 4;
}
if let (Some(fi), Some((di, sz))) = (find(slb, b"fmt ", 0), best) {
if sz > 512 {
let fsz = le32(slb, fi + 4)? as usize;
let fmt_end = (fi + 8 + fsz).min(slb.len());
let data = slb.get(di + 8..di + 8 + sz)?;
return Some(build_riff(slb.get(fi..fmt_end)?, data));
}
}
// Headerless fallback: fixed data offset, synthesized XMA1 stereo/48k fmt.
let data = slb.get(HEADERLESS_DATA_OFFSET..)?;
(!data.is_empty()).then(|| build_riff(&synth_xma1_fmt(2, 2, 48000), data))
}
/// A minimal `fmt ` chunk carrying an XMA1 `XMAWAVEFORMAT` (one stream). /// A minimal `fmt ` chunk carrying an XMA1 `XMAWAVEFORMAT` (one stream).
fn synth_xma1_fmt(channels: u8, channel_mask: u16, rate: u32) -> Vec<u8> { fn synth_xma1_fmt(channels: u8, channel_mask: u16, rate: u32) -> Vec<u8> {
let mut fmt = Vec::with_capacity(40); let mut fmt = Vec::with_capacity(40);

View File

@@ -1752,9 +1752,13 @@ fn poll_loader_channel(
generation, generation,
wav, wav,
}) => { }) => {
if generation == mvoice.generation let accept = generation == mvoice.generation
&& mvoice.movie.as_deref() == Some(movie.as_str()) && mvoice.movie.as_deref() == Some(movie.as_str());
{ info!(
"[voice] loaded movie={movie:?} gen={generation} (cur movie={:?} gen={}) accept={accept} wav={wav:?}",
mvoice.movie, mvoice.generation
);
if accept {
mvoice.loading = false; mvoice.loading = false;
mvoice.available = wav.is_some(); mvoice.available = wav.is_some();
mvoice.pending_wav = wav; mvoice.pending_wav = wav;
@@ -3140,30 +3144,63 @@ fn decode_sound_clip(
use sylpheed_formats::{hash::name_hash, slb}; use sylpheed_formats::{hash::name_hash, slb};
let key = name_hash(clip_name); let key = name_hash(clip_name);
let bytes = read_sound_entry(source, key)?; let bytes = read_sound_entry(source, key)?;
// Decode the first sub-wave (base behaviour). NOTE: multi-sub-wave assembly // A `.slb` is an XACT bank of one or more sub-waves. Decode EVERY sub-wave and
// (segments vs alternate takes vs timecode-placed lines) is UNRESOLVED — a // concatenate them, then clamp to the movie length. This is robust to both bank
// naive concat produced the wrong track for RT movies. Pending ground-truth // shapes we see: segment banks (e.g. VOICE_RT07A = 3 sub-waves 24+14+11s ≈ the
// voice-playback instrumentation (Canary APU/XMA) before a real fix. // 50s movie) yield the full dialogue, while the clamp drops the extra ALTERNATE
let riff = slb::to_xma_riff(&bytes).ok_or("not a decodable .slb")?; // takes of full-length banks (e.g. VOICE_S00A, whose sub-wave 0 already spans
// the 94s movie). Playing only sub-wave 0 (the old behaviour) dropped 2/3 of the
// dialogue for segment banks — the "RT voice wrong" bug. Verified vs real durations.
let mut riffs = slb::to_xma_riffs(&bytes);
if riffs.is_empty() {
// Some banks (data-before-header `\etc\` radio clips) confuse the
// multi-sub-wave scanner; the robust single-stream decoder handles them,
// so the standalone browser can still play them.
riffs = slb::to_xma_riff_best(&bytes).into_iter().collect();
}
decode_riffs_to_wav(riffs, &format!("{key:08x}"), duration)
}
/// Decode a set of XMA sub-wave RIFFs into a single mono WAV: concatenate them,
/// downmix to mono (left channel holds the signal for dual-mono sources), and
/// clamp to `duration` seconds. Shared by the per-`.slb` decoder and the
/// continuous movie-voice decoder.
#[cfg(not(target_arch = "wasm32"))]
fn decode_riffs_to_wav(riffs: Vec<Vec<u8>>, tag: &str, duration: f32) -> Result<PathBuf, String> {
if riffs.is_empty() {
return Err("no decodable audio".into());
}
let dir = std::env::temp_dir(); let dir = std::env::temp_dir();
let xma_path = dir.join(format!("sylph_voice_{key:08x}.xma.wav")); let out_path = dir.join(format!("sylph_voice_{tag}.wav"));
let out_path = dir.join(format!("sylph_voice_{key:08x}.wav")); let mut inputs = Vec::with_capacity(riffs.len());
std::fs::write(&xma_path, &riff).map_err(|e| e.to_string())?; for (i, riff) in riffs.iter().enumerate() {
let p = dir.join(format!("sylph_voice_{tag}_{i}.xma.wav"));
std::fs::write(&p, riff).map_err(|e| e.to_string())?;
inputs.push(p);
}
let dur = if duration.is_finite() && duration > 0.0 { let dur = if duration.is_finite() && duration > 0.0 {
duration duration
} else { } else {
600.0 600.0
}; };
let status = Command::new("ffmpeg") let filter = format!(
.args(["-hide_banner", "-y", "-i"]) "{}concat=n={}:v=0:a=1,pan=mono|c0=c0[a]",
.arg(&xma_path) (0..inputs.len()).map(|i| format!("[{i}:a]")).collect::<String>(),
.args(["-af", "pan=mono|c0=c0", "-t"]) inputs.len(),
);
let mut cmd = Command::new("ffmpeg");
cmd.args(["-hide_banner", "-y"]);
for p in &inputs {
cmd.arg("-i").arg(p);
}
cmd.args(["-filter_complex", &filter, "-map", "[a]", "-t"])
.arg(format!("{dur}")) .arg(format!("{dur}"))
.arg(&out_path) .arg(&out_path);
.output(); let status = cmd.output();
let _ = std::fs::remove_file(&xma_path); for p in &inputs {
let _ = std::fs::remove_file(p);
}
match status { match status {
Ok(o) if o.status.success() && out_path.metadata().map(|m| m.len() > 44).unwrap_or(false) => { Ok(o) if o.status.success() && out_path.metadata().map(|m| m.len() > 44).unwrap_or(false) => {
Ok(out_path) Ok(out_path)
@@ -3176,6 +3213,115 @@ fn decode_sound_clip(
} }
} }
/// Categorical fallback voice token for a hokyu (resupply) cutscene the manifest
/// leaves unbound. Only 5 of the 18 hokyu movies carry an explicit `VOICETRACK`;
/// the game reuses those 5 generic recordings across stages, keyed by
/// ship (`LS`/`DS`) × resupply source (carrier `…A` / tanker `…H`). The four
/// category→recording pairs are read off the bound examples; `LS`/carrier has two
/// takes (450 from s02A, 451 from s09A) — we default unbound LS/carrier to 450.
/// Returns `None` for non-hokyu movies (they resolve via the manifest only).
fn hokyu_fallback_token(movie: &str) -> Option<String> {
if !movie.starts_with("hokyu_") {
return None;
}
let ls = movie.contains("_LS_");
let ds = movie.contains("_DS_");
let carrier = movie.ends_with('A');
let tanker = movie.ends_with('H');
let token = match (ls, ds, carrier, tanker) {
(true, _, true, _) => "VOICE_D_450", // light ship, carrier resupply
(true, _, _, true) => "VOICE_D_453", // light ship, tanker resupply
(_, true, true, _) => "VOICE_D_452", // Delta Saber, carrier resupply
(_, true, _, true) => "VOICE_D_454", // Delta Saber, tanker resupply
_ => return None,
};
Some(token.to_string())
}
/// Resolve a movie's cutscene voice to a **continuous byte region** of the
/// `sound.pNN` voice stream, in `[start, end)` global offsets.
///
/// The movie voices are one continuous XMA stream chunked into `VOICE_*.slb` TOC
/// entries whose boundaries do NOT match the cutscene cues (a cue routinely spans
/// two `.slb` chunks — so a `.slb` need not hold the track its name claims). Each
/// cue ends at an inline `(sound_id, 0x11, …)` trailer; cue N = the bytes between
/// trailer N-1 and trailer N. Chain: movie → cue name (manifest VOICETRACK) →
/// sound-id (master registry) → region (scan the stream for the two trailers).
/// Returns `None` for movies whose voice is not a `\Movie\` bank (e.g. hokyu
/// `\etc\` clips), which the caller then resolves the legacy per-clip way.
#[cfg(not(target_arch = "wasm32"))]
fn resolve_movie_voice_region(
source: &SourceKind,
movie: &str,
lang: sylpheed_formats::slb::VoiceLang,
) -> Option<(u64, u64)> {
use sylpheed_formats::{hash::name_hash, movie_manifest, movie_voice, PakArchive};
let code = lang.code_pub();
let tpak = read_pak_archive_blocking(source, "dat/tables.pak").ok()?;
// movie → cue token (e.g. "VOICE_RT01A")
let manifest = tpak
.entries()
.iter()
.find_map(|e| tpak.read(e).ok().filter(|b| movie_manifest::is_manifest(b)))?;
let token = movie_manifest::voice_token(&manifest, movie)
.or_else(|| hokyu_fallback_token(movie))?;
// token → sound-id via the master registry (the large per-language IDXD entry
// carrying the `<lang>\Movie\VOICE_*.slb` paths).
let marker = format!("{code}\\Movie\\VOICE_ADV.slb");
let registry = tpak.entries().iter().find_map(|e| {
tpak.read(e)
.ok()
.filter(|b| b.windows(marker.len()).any(|w| w == marker.as_bytes()))
})?;
let id = *movie_voice::registry_voice_ids(&registry).get(&token)?;
// physical anchor: the TOC offset of this token's own `.slb` chunk — a start
// point near the cue's trailers (the cue itself may sit before/after it). The
// token's subdir varies: `Movie` (ADV/RT/S cutscenes) or `etc` (hokyu VOICE_D).
let stoc = read_source_file(source, "dat/sound.pak").ok()?;
let entries = PakArchive::parse_toc(&stoc).ok()?;
let anchor = ["Movie", "etc", "Voice"].iter().find_map(|dir| {
let h = name_hash(&format!("{code}\\{dir}\\{token}.slb"));
entries
.binary_search_by_key(&h, |e| e.name_hash)
.ok()
.map(|i| entries[i].offset as u64)
})?;
// Scan a window spanning both directions from the anchor for this cue's trailer
// and its predecessor. The window must cover the largest bank (ADV ≈ 3.6 MB).
let win_start = anchor.saturating_sub(2 * 1024 * 1024) & !3;
let window = read_segment_range(source, "dat/sound", win_start, 8 * 1024 * 1024).ok()?;
let end_local = movie_voice::find_descriptor(&window, id)?;
let end = win_start + end_local as u64;
// Cue start = its predecessor trailer. Prefer the exact `id-1` trailer; if the
// id sequence has a gap (e.g. VOICE_D_453→454), fall back to the nearest trailer
// below this one — but only if it's within one bank (~1.5 MB), else this is the
// first cue in its block and the audio starts at the bank anchor itself.
let start = movie_voice::find_descriptor(&window, id.wrapping_sub(1))
.or_else(|| movie_voice::find_descriptor_before(&window, end_local))
.map(|o| win_start + o as u64)
.filter(|&s| s < end && end - s < 1_500_000)
.unwrap_or(anchor);
Some((start, end))
}
/// Decode a continuous movie-voice byte region (from [`resolve_movie_voice_region`])
/// into a mono WAV, clamped to `duration`.
#[cfg(not(target_arch = "wasm32"))]
fn decode_voice_region(
source: &SourceKind,
start: u64,
end: u64,
duration: f32,
) -> Result<PathBuf, String> {
use sylpheed_formats::slb;
let bytes = read_segment_range(source, "dat/sound", start, (end - start) as usize)?;
let mut riffs = slb::to_xma_riffs(&bytes);
if riffs.is_empty() {
riffs = slb::to_xma_riff_best(&bytes).into_iter().collect();
}
decode_riffs_to_wav(riffs, &format!("mv_{start:x}"), duration)
}
/// Resolve a movie's `sound.pak` voice-entry name via the **movie manifest** /// Resolve a movie's `sound.pak` voice-entry name via the **movie manifest**
/// (`tables.pak`), which authoritatively binds each movie to its voice bank — /// (`tables.pak`), which authoritatively binds each movie to its voice bank —
/// several `hokyu_*` resupply movies point at in-mission `VOICE_D_*` clips in /// several `hokyu_*` resupply movies point at in-mission `VOICE_D_*` clips in
@@ -3221,8 +3367,24 @@ fn handle_voice_request(
let (movie, lang, duration, generation) = let (movie, lang, duration, generation) =
(req.movie.clone(), req.lang, req.duration, req.generation); (req.movie.clone(), req.lang, req.duration, req.generation);
std::thread::spawn(move || { std::thread::spawn(move || {
let wav = resolve_movie_voice_clip(&source, &movie, lang) // The cutscene voices live in one continuous XMA stream, addressed by cue
.and_then(|clip| decode_sound_clip(&source, &clip, duration).ok()); // byte-region (movie → sound-id → [trailer(N-1)..trailer(N)]); this gives
// the CORRECT, complete voice for ADV / S / RT movies, including the many
// cues that span two `.slb` chunks. Movies whose voice is not a `\Movie\`
// bank (hokyu resupply → `\etc\` demo clips) resolve to no region and fall
// through to the legacy per-clip decoder. A `\Movie\` token that failed
// region resolution must NOT play its raw `.slb` — that off-by-one chunk is
// the wrong track — so it stays silent rather than wrong.
let wav = resolve_movie_voice_region(&source, &movie, lang)
.and_then(|(s, e)| decode_voice_region(&source, s, e, duration).ok())
.or_else(|| {
let clip = resolve_movie_voice_clip(&source, &movie, lang)?;
if clip.contains("\\Movie\\") {
return None;
}
decode_sound_clip(&source, &clip, duration).ok()
});
info!("[voice] movie={movie:?} gen={generation} -> wav={wav:?}");
let _ = sender.send(IsoLoaderMsg::VoiceLoaded { let _ = sender.send(IsoLoaderMsg::VoiceLoaded {
movie, movie,
generation, generation,
@@ -3265,15 +3427,22 @@ fn handle_audio_request(
req.generation, req.generation,
); );
std::thread::spawn(move || { std::thread::spawn(move || {
// Resolve the movie voice bank from the manifest when asked, else decode // For a movie, decode its full-length cutscene voice via the continuous
// the named clip directly. // stream (same resolution as playback), falling back to a non-`\Movie\`
let entry = match movie { // clip (hokyu `\etc\`); for a named library clip, decode it directly.
Some((m, lang)) => resolve_movie_voice_clip(&source, &m, lang), let wav = match movie {
None => Some(clip), Some((m, lang)) => resolve_movie_voice_region(&source, &m, lang)
}; .and_then(|(s, e)| decode_voice_region(&source, s, e, f32::INFINITY).ok())
let wav = entry .or_else(|| {
.and_then(|c| decode_sound_clip(&source, &c, f32::INFINITY).ok()) let c = resolve_movie_voice_clip(&source, &m, lang)?;
.and_then(|p| wav_duration(&p).map(|d| (p, d))); if c.contains("\\Movie\\") {
return None;
}
decode_sound_clip(&source, &c, f32::INFINITY).ok()
}),
None => decode_sound_clip(&source, &clip, f32::INFINITY).ok(),
}
.and_then(|p| wav_duration(&p).map(|d| (p, d)));
let _ = sender.send(IsoLoaderMsg::AudioLoaded { let _ = sender.send(IsoLoaderMsg::AudioLoaded {
name: display, name: display,
generation, generation,
@@ -3414,6 +3583,7 @@ fn advance_video_playback(
if let Some(wav) = voice.pending_wav.take() { if let Some(wav) = voice.pending_wav.take() {
if let Some(handle) = &inner.stream_handle { if let Some(handle) = &inner.stream_handle {
let vol = if voice.enabled { video.volume } else { 0.0 }; let vol = if voice.enabled { video.volume } else { 0.0 };
info!("[voice] APPLY sink movie={:?} wav={wav:?} pos={}", voice.movie, video.position);
inner.voice_sink = inner.voice_sink =
build_audio_sink(handle, &wav, video.position, vol, video.playing && !video.scrubbing); build_audio_sink(handle, &wav, video.position, vol, video.playing && !video.scrubbing);
inner.voice_applied_volume = vol; inner.voice_applied_volume = vol;

View File

@@ -227,37 +227,73 @@ fn draw_viewer_ui(
ui.label("Filter:"); ui.label("Filter:");
ui.text_edit_singleline(&mut voice_lib.filter); ui.text_edit_singleline(&mut voice_lib.filter);
}); });
let f = voice_lib.filter.to_lowercase();
// Group the thousands of entries as directory → speaker so the
// list is navigable (e.g. browse `Voice` by character to find a
// cutscene's radio line). `name` is `<lang>\<dir>\<file>.slb`.
use std::collections::BTreeMap;
let mut groups: BTreeMap<&str, BTreeMap<&str, Vec<&sylpheed_formats::slb::VoiceClip>>> =
BTreeMap::new();
for c in &voice_lib.clips {
if !f.is_empty() && !c.name.to_lowercase().contains(&f) {
continue;
}
let dir = c.name.rsplit('\\').nth(1).unwrap_or("?");
groups
.entry(dir)
.or_default()
.entry(c.speaker.as_str())
.or_default()
.push(c);
}
let shown: usize = groups.values().flat_map(|s| s.values()).map(Vec::len).sum();
ui.label( ui.label(
egui::RichText::new(format!("{} clips", voice_lib.clips.len())) egui::RichText::new(format!("{shown} / {} clips", voice_lib.clips.len()))
.weak() .weak()
.small(), .small(),
); );
ui.separator(); ui.separator();
let f = voice_lib.filter.to_lowercase(); let filtering = !f.is_empty();
let clips: Vec<_> = voice_lib
.clips
.iter()
.filter(|c| f.is_empty() || c.name.to_lowercase().contains(&f))
.take(2000)
.cloned()
.collect();
egui::ScrollArea::vertical().show(ui, |ui| { egui::ScrollArea::vertical().show(ui, |ui| {
for c in &clips { for (dir, speakers) in &groups {
ui.horizontal(|ui| { let dtotal: usize = speakers.values().map(Vec::len).sum();
if ui.button("").on_hover_text(&c.name).clicked() { egui::CollapsingHeader::new(format!("📁 {dir} ({dtotal})"))
audio.generation = audio.generation.wrapping_add(1); .id_salt(("vdir", *dir))
audio.loading = true; .default_open(filtering)
audio.active = true; .show(ui, |ui| {
audio.name = c.display.clone(); for (speaker, clips) in speakers {
events.audio.send(RequestAudio { egui::CollapsingHeader::new(format!(
clip: c.name.clone(), "{speaker} ({})",
display: c.display.clone(), clips.len()
movie: None, ))
generation: audio.generation, .id_salt(("vspk", *dir, *speaker))
}); .default_open(filtering || clips.len() <= 6)
} .show(ui, |ui| {
ui.label(&c.display); for c in clips {
}); ui.horizontal(|ui| {
if ui
.button("")
.on_hover_text(&c.name)
.clicked()
{
audio.generation =
audio.generation.wrapping_add(1);
audio.loading = true;
audio.active = true;
audio.name = c.display.clone();
events.audio.send(RequestAudio {
clip: c.name.clone(),
display: c.display.clone(),
movie: None,
generation: audio.generation,
});
}
ui.label(&c.display);
});
}
});
}
});
} }
}); });
} }