diff --git a/crates/apr-cli/src/commands/bench.rs b/crates/apr-cli/src/commands/bench.rs index 94a0a7bd43..271f214919 100644 --- a/crates/apr-cli/src/commands/bench.rs +++ b/crates/apr-cli/src/commands/bench.rs @@ -828,6 +828,10 @@ fn print_header(path: &Path, config: &BenchConfig) { } include!("benchmark.rs"); + +#[cfg(all(test, feature = "inference", target_os = "linux"))] +#[path = "bench_rss_tests.rs"] +mod rss_tests; include!("bench_safetensors.rs"); include!("bench_moe.rs"); include!("bench_qwen35.rs"); diff --git a/crates/apr-cli/src/commands/bench_moe.rs b/crates/apr-cli/src/commands/bench_moe.rs index 72f93902e9..71085209a7 100644 --- a/crates/apr-cli/src/commands/bench_moe.rs +++ b/crates/apr-cli/src/commands/bench_moe.rs @@ -31,24 +31,21 @@ fn is_moe_gguf(gguf: &realizar::gguf::GGUFModel) -> bool { /// then runs warmup + iterations through the appropriate forward path. #[cfg(feature = "inference")] fn run_gguf_moe_benchmark( - path: &Path, + mapped: &realizar::gguf::MappedGGUFModel, config: &BenchConfig, use_cuda: bool, prompt_tokens: &[u32], + start: Instant, _tracer: &TracerImpl, ) -> Result { use realizar::gguf::qwen3_moe_load::load_qwen3_moe_layer; - use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel}; + use realizar::gguf::OwnedQuantizedModel; if !config.quiet { eprintln!("{}", "Loading MoE GGUF model...".yellow()); } - let start = Instant::now(); - let mapped = MappedGGUFModel::from_path(path) - .map_err(|e| CliError::ValidationFailed(format!("Failed to mmap MoE model: {e}")))?; - - let model = OwnedQuantizedModel::from_mapped(&mapped) + let model = OwnedQuantizedModel::from_mapped(mapped) .map_err(|e| CliError::ValidationFailed(format!("Failed to create MoE model: {e}")))?; // Read MoE config from GGUF metadata. expert_count() must be Some diff --git a/crates/apr-cli/src/commands/bench_rss_tests.rs b/crates/apr-cli/src/commands/bench_rss_tests.rs new file mode 100644 index 0000000000..bed1a13260 --- /dev/null +++ b/crates/apr-cli/src/commands/bench_rss_tests.rs @@ -0,0 +1,51 @@ +//! #3761 case row: `apr bench` on a GGUF detects the format from 8 bytes and maps the model +//! ONCE. The bench runs the model, so the map is its floor: realizar's map pre-faults every page +//! (MAP_POPULATE, PMAT-304), which puts the whole file in RSS. The row proves nothing is read on +//! top of that one map: it read the whole file onto the heap first, and each backend path then +//! mapped it again. The fixture's header parses and its model does not build (one tensor), so +//! the run stops after the reads under test. +//! +//! Measured (x86-64 debug, 2026-09-21): 2,123,680-2,125,672 KiB over three runs, i.e. the +//! 2,097,152 KiB map and about 27 MiB. The whole-file read put back into the format check, or +//! into the tokenizer's parse, gives 4,212,352 / 4,212,152 KiB, and so did the second map this +//! change removed (4,224,696 KiB). + +use super::*; +use crate::commands::model_header::rss_probe; + +#[test] +fn bench_of_a_2_gib_gguf_keeps_peak_rss_small() { + let f = rss_probe::sparse( + &rss_probe::gguf_with_vocab(), + rss_probe::FIXTURE_LEN, + ".gguf", + ); + let hwm = rss_probe::child_peak_kb( + "commands::bench::rss_tests::bench_peak_rss_probe", + &[(BENCH_PROBE, f.path())], + ); + // One map of the file, plus the margin every header row gets. + let bound = rss_probe::FIXTURE_LEN / 1024 + rss_probe::PEAK_RSS_BOUND_KB; + assert!( + hwm < bound, + "peak RSS {hwm} KiB benchmarking a 2 GiB GGUF (bound {bound} KiB, one map + margin): \ + a whole-file read or a second map is back" + ); +} + +const BENCH_PROBE: &str = "APR_3761_BENCH_PROBE"; + +/// Not a test on its own: with the probe variable unset it does nothing. +#[test] +fn bench_peak_rss_probe() { + let Some(path) = std::env::var_os(BENCH_PROBE) else { + return; + }; + let config = BenchConfig { + quiet: true, + ..BenchConfig::default() + }; + let result = run_realizar_benchmark(Path::new(&path), &config); + assert!(result.is_err(), "a one-tensor GGUF cannot be benchmarked"); + rss_probe::report_peak(); +} diff --git a/crates/apr-cli/src/commands/bench_safetensors.rs b/crates/apr-cli/src/commands/bench_safetensors.rs index 2cd2adf28e..8f516d89af 100644 --- a/crates/apr-cli/src/commands/bench_safetensors.rs +++ b/crates/apr-cli/src/commands/bench_safetensors.rs @@ -273,25 +273,20 @@ fn run_cuda_measurement( #[cfg_attr(coverage_nightly, coverage(off))] #[cfg(all(feature = "inference", feature = "cuda"))] fn run_cuda_benchmark( - _gguf: &realizar::gguf::GGUFModel, - _model_bytes: &[u8], + mapped: &realizar::gguf::MappedGGUFModel, prompt_tokens: &[u32], gen_config: &realizar::gguf::QuantizedGenerateConfig, config: &BenchConfig, start: Instant, - model_path: &Path, tracer: &TracerImpl, ) -> Result { - use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel, OwnedQuantizedModelCuda}; + use realizar::gguf::{OwnedQuantizedModel, OwnedQuantizedModelCuda}; if !config.quiet { eprintln!("{}", "Initializing CUDA model...".cyan()); } - let mapped = MappedGGUFModel::from_path(model_path) - .map_err(|e| CliError::ValidationFailed(format!("Failed to map model: {e}")))?; - - let model = OwnedQuantizedModel::from_mapped(&mapped) + let model = OwnedQuantizedModel::from_mapped(mapped) .map_err(|e| CliError::ValidationFailed(format!("Failed to create model: {e}")))?; let mut cuda_model = OwnedQuantizedModelCuda::new(model, 0) @@ -318,18 +313,16 @@ fn run_cuda_benchmark( /// CPU-based benchmark fallback path #[cfg(feature = "inference")] fn run_cpu_benchmark( + mapped: &realizar::gguf::MappedGGUFModel, prompt_tokens: &[u32], gen_config: &realizar::gguf::QuantizedGenerateConfig, config: &BenchConfig, start: Instant, - path: &Path, tracer: &TracerImpl, ) -> Result { - use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel}; + use realizar::gguf::OwnedQuantizedModel; - let mapped = MappedGGUFModel::from_path(path) - .map_err(|e| CliError::ValidationFailed(format!("Failed to mmap model: {e}")))?; - let model = OwnedQuantizedModel::from_mapped(&mapped) + let model = OwnedQuantizedModel::from_mapped(mapped) .map_err(|e| CliError::ValidationFailed(format!("Failed to create model: {e}")))?; bench_log_ready(config, start.elapsed(), " (CPU)"); diff --git a/crates/apr-cli/src/commands/benchmark.rs b/crates/apr-cli/src/commands/benchmark.rs index 610d41c489..f925a684ce 100644 --- a/crates/apr-cli/src/commands/benchmark.rs +++ b/crates/apr-cli/src/commands/benchmark.rs @@ -100,12 +100,12 @@ fn print_results(result: &BenchResult) { fn run_realizar_benchmark(path: &Path, config: &BenchConfig) -> Result { use realizar::format::{detect_format, ModelFormat}; - // Read first 8 bytes for format detection - let header_bytes = std::fs::read(path) + // Read first 8 bytes for format detection (#3761: it read the whole model under this comment) + let header_bytes = super::model_header::read_prefix(path, 8) .map_err(|e| CliError::ValidationFailed(format!("Failed to read model: {e}")))?; // Detect format - let format = detect_format(&header_bytes[..8.min(header_bytes.len())]) + let format = detect_format(&header_bytes) .map_err(|e| CliError::ValidationFailed(format!("Failed to detect format: {e}")))?; if !config.quiet { @@ -156,18 +156,19 @@ fn run_gguf_benchmark( use_cuda: bool, tracer: &TracerImpl, ) -> Result { - use realizar::gguf::{GGUFModel, QuantizedGenerateConfig}; + use realizar::gguf::{MappedGGUFModel, QuantizedGenerateConfig}; if !config.quiet { eprintln!("{}", "Loading GGUF model...".yellow()); } let start = Instant::now(); - // Load model for tokenization - let model_bytes = std::fs::read(path) - .map_err(|e| CliError::ValidationFailed(format!("Failed to read model: {e}")))?; - let gguf = GGUFModel::from_bytes(&model_bytes) - .map_err(|e| CliError::ValidationFailed(format!("Failed to parse GGUF: {e}")))?; + // The ONE map of the model (#3761). The tokenizer and the MoE fields come from its header, + // and inference runs on it. This read the whole file first, and each path then mapped it + // again: a map pre-faults every page (MAP_POPULATE), so two maps doubled the peak. + let mapped = MappedGGUFModel::from_path(path) + .map_err(|e| CliError::ValidationFailed(format!("Failed to map GGUF: {e}")))?; + let gguf = &mapped.model; // Tokenize prompt let bos = aprender::demo::SpecialTokens::qwen2().bos_id; @@ -212,7 +213,7 @@ fn run_gguf_benchmark( gguf.expert_used_count().unwrap_or(0) ); } - return run_gguf_moe_benchmark(path, config, use_cuda, &prompt_tokens, tracer); + return run_gguf_moe_benchmark(&mapped, config, use_cuda, &prompt_tokens, start, tracer); } let gen_config = QuantizedGenerateConfig { @@ -224,16 +225,7 @@ fn run_gguf_benchmark( #[cfg(feature = "cuda")] if use_cuda { - match run_cuda_benchmark( - &gguf, - &model_bytes, - &prompt_tokens, - &gen_config, - config, - start, - path, - tracer, - ) { + match run_cuda_benchmark(&mapped, &prompt_tokens, &gen_config, config, start, tracer) { Ok(result) => return Ok(result), Err(e) => { // GH-284: Fall back to CPU on CUDA capability mismatch (e.g. missing QkNorm kernel) @@ -245,17 +237,17 @@ fn run_gguf_benchmark( } let cpu_start = Instant::now(); return run_cpu_benchmark( + &mapped, &prompt_tokens, &gen_config, config, cpu_start, - path, tracer, ); } } } - run_cpu_benchmark(&prompt_tokens, &gen_config, config, start, path, tracer) + run_cpu_benchmark(&mapped, &prompt_tokens, &gen_config, config, start, tracer) } /// Resolve prompt tokens from APR model's tokenizer, with fallbacks. diff --git a/crates/apr-cli/src/commands/embed_viz.rs b/crates/apr-cli/src/commands/embed_viz.rs index d2eb0122b0..0a76e9ac8c 100644 --- a/crates/apr-cli/src/commands/embed_viz.rs +++ b/crates/apr-cli/src/commands/embed_viz.rs @@ -379,10 +379,11 @@ fn resolve_tokens( /// read ONCE and can also ask it how many tokens the model declares — the /// cross-check in `check_vocab_axis`. fn gguf_vocab(model: &Path) -> Option { - let bytes = std::fs::read(model).ok()?; - if !bytes.starts_with(b"GGUF") { + // #3761: the complete header only (the vocabulary lives there), never the tensor data + if super::model_header::read_prefix(model, 4).ok()?.as_slice() != b"GGUF" { return None; } + let bytes = super::model_header::gguf_header_bytes(model).ok()?; LlamaTokenizer::from_gguf_bytes(&bytes).ok() } diff --git a/crates/apr-cli/src/commands/embed_viz_tests.rs b/crates/apr-cli/src/commands/embed_viz_tests.rs index 25bea59470..d769747a5e 100644 --- a/crates/apr-cli/src/commands/embed_viz_tests.rs +++ b/crates/apr-cli/src/commands/embed_viz_tests.rs @@ -595,3 +595,42 @@ fn pca_needs_at_least_two_rows() { let err = run(&a).expect_err("PCA on one sample has no variance to decompose"); assert!(err.to_string().contains("at least 2 rows"), "got: {err}"); } + +/// #3761 case row: `gguf_vocab` reads a 2 GiB GGUF's header (where the vocabulary lives) and +/// never its tensor data. Measured (x86-64 debug, 2026-09-21): 39.8-42.3 MiB over three runs; with `gguf_vocab` reading the +/// whole file again, 2,116,644 KiB, and with `gguf_header_bytes` doing so, 2,117,924 KiB. +#[cfg(target_os = "linux")] +#[test] +fn gguf_vocab_of_a_2_gib_gguf_keeps_peak_rss_small() { + use crate::commands::model_header::rss_probe; + let f = rss_probe::sparse( + &rss_probe::gguf_with_vocab(), + rss_probe::FIXTURE_LEN, + ".gguf", + ); + let hwm = rss_probe::child_peak_kb( + "commands::embed_viz::tests::gguf_vocab_peak_rss_probe", + &[(VOCAB_PROBE, f.path())], + ); + assert!( + hwm < rss_probe::PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB reading the vocabulary of a 2 GiB GGUF (bound {} KiB): a whole-file read is back", + rss_probe::PEAK_RSS_BOUND_KB + ); +} + +#[cfg(target_os = "linux")] +const VOCAB_PROBE: &str = "APR_3761_VOCAB_PROBE"; + +/// Not a test on its own: with the probe variable unset it does nothing. +#[cfg(target_os = "linux")] +#[test] +fn gguf_vocab_peak_rss_probe() { + let Some(path) = std::env::var_os(VOCAB_PROBE) else { + return; + }; + let tokenizer = + gguf_vocab(std::path::Path::new(&path)).expect("the vocabulary, from the header"); + assert_eq!(tokenizer.vocab_size(), 4); + crate::commands::model_header::rss_probe::report_peak(); +} diff --git a/crates/apr-cli/src/commands/eval/eval_mod_tests.rs b/crates/apr-cli/src/commands/eval/eval_mod_tests.rs index 27eec3c9f6..adf61ae7af 100644 --- a/crates/apr-cli/src/commands/eval/eval_mod_tests.rs +++ b/crates/apr-cli/src/commands/eval/eval_mod_tests.rs @@ -108,6 +108,26 @@ fn format_archive_size_units() { assert_eq!(format_archive_size(1_610_612_736), "1.5 GB"); } +// ── compute_file_hash_streamed (#3761): the same hash, chunk by chunk ────── + +#[test] +fn streamed_file_hash_equals_the_whole_buffer_hash() { + use std::io::Write; + // 2.5 MiB, so the 1 MiB chunks split it three ways, one part partial + let bytes: Vec = (0..(5usize << 19)).map(|i| (i * 31 % 251) as u8).collect(); + let mut f = tempfile::NamedTempFile::new().expect("temp file"); + f.write_all(&bytes).expect("write"); + assert_eq!( + compute_file_hash_streamed(f.path()).expect("stream the hash"), + compute_file_hash(&bytes) + ); + let empty = tempfile::NamedTempFile::new().expect("temp file"); + assert_eq!( + compute_file_hash_streamed(empty.path()).expect("empty"), + "cbf29ce484222325" + ); +} + // ── compute_file_hash (FNV-1a) ───────────────────────────────────────────── #[test] @@ -479,3 +499,57 @@ fn default_prompts_are_nonempty_python() { assert!(p.contains("def ") || p.contains("class ")); } } + +/// #3761 case row: `count_safetensors_keys` reads a 2 GiB SafeTensors file's JSON header, and +/// `verify_single_file` hashes every byte of it STREAMED; neither holds the file in memory. +/// Measured (x86-64 debug, 2026-09-21): 25.3-27.3 MiB over three runs. A whole-file read put back +/// into `count_safetensors_keys`, into `verify_single_file`'s header read, or into its hash: +/// 2,115,124 / 2,114,024 / 2,114,080 KiB. +#[cfg(target_os = "linux")] +#[test] +fn safetensors_readers_of_a_2_gib_file_keep_peak_rss_small() { + use crate::commands::model_header::rss_probe; + let f = rss_probe::sparse( + &rss_probe::safetensors(), + rss_probe::FIXTURE_LEN, + ".safetensors", + ); + let hwm = rss_probe::child_peak_kb( + "commands::eval::eval_mod_tests::safetensors_readers_peak_rss_probe", + &[(EVAL_PROBE, f.path())], + ); + assert!( + hwm < rss_probe::PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB on a 2 GiB SafeTensors file (bound {} KiB): a whole-file read is back", + rss_probe::PEAK_RSS_BOUND_KB + ); +} + +#[cfg(target_os = "linux")] +const EVAL_PROBE: &str = "APR_3761_EVAL_PROBE"; + +/// Not a test on its own: with the probe variable unset it does nothing. +#[cfg(target_os = "linux")] +#[test] +fn safetensors_readers_peak_rss_probe() { + let Some(path) = std::env::var_os(EVAL_PROBE) else { + return; + }; + let path = std::path::Path::new(&path); + assert_eq!(count_safetensors_keys(path), 1); + let mut checks = Vec::new(); + verify_single_file(path, &mut checks).expect("verify"); + assert!( + checks + .iter() + .any(|(name, ok)| name == "1 tensors found" && *ok), + "{checks:?}" + ); + assert!( + checks + .iter() + .any(|(name, _)| name.starts_with("BLAKE3 hash: ")), + "{checks:?}" + ); + crate::commands::model_header::rss_probe::report_peak(); +} diff --git a/crates/apr-cli/src/commands/eval/mod.rs b/crates/apr-cli/src/commands/eval/mod.rs index 49660d6935..bddee422dd 100644 --- a/crates/apr-cli/src/commands/eval/mod.rs +++ b/crates/apr-cli/src/commands/eval/mod.rs @@ -743,7 +743,8 @@ fn gather_model_info(path: &Path) -> Result { /// Count tensor keys in a safetensors file by reading the header. fn count_safetensors_keys(path: &Path) -> usize { - let Ok(data) = std::fs::read(path) else { + // #3761: the length and the JSON header only, under the shared prefix policy + let Ok(data) = aprender::format::prefix::safetensors_header_prefix(path) else { return 0; }; if data.len() < 8 { @@ -919,21 +920,32 @@ fn verify_single_file(path: &Path, checks: &mut Vec<(String, bool)>) -> Result<( // safetensors format check if path.extension().is_some_and(|e| e == "safetensors") { - let data = std::fs::read(path).map_err(|e| { + // #3761: the 8-byte length, then the JSON header when its size is valid, and a STREAMED + // hash of every byte. The whole file is never held in memory. + let cannot_read = |e: std::io::Error| { CliError::ValidationFailed(format!("Cannot read {}: {e}", path.display())) - })?; + }; + let head = super::model_header::read_prefix(path, 8).map_err(cannot_read)?; // Valid header - let header_ok = data.len() >= 8; + let header_ok = head.len() >= 8; checks.push(("safetensors header present".to_string(), header_ok)); if header_ok { - let header_size = u64::from_le_bytes(data[..8].try_into().unwrap_or_default()) as usize; - let header_valid = data.len() >= 8 + header_size && header_size < 100_000_000; + let header_size = u64::from_le_bytes(head[..8].try_into().unwrap_or_default()) as usize; + let header_valid = (header_size as u64) + .checked_add(8) + .is_some_and(|n| metadata.len() >= n) + && header_size < 100_000_000; checks.push(("safetensors header valid size".to_string(), header_valid)); if header_valid { - let header_str = std::str::from_utf8(&data[8..8 + header_size]).unwrap_or(""); + let data = + super::model_header::read_prefix(path, 8 + header_size).map_err(cannot_read)?; + let header_str = data + .get(8..8 + header_size) + .and_then(|b| std::str::from_utf8(b).ok()) + .unwrap_or(""); let header_json = serde_json::from_str::(header_str).is_ok(); checks.push(("safetensors header valid JSON".to_string(), header_json)); @@ -946,8 +958,8 @@ fn verify_single_file(path: &Path, checks: &mut Vec<(String, bool)>) -> Result<( } } - // Hash check: compute simple checksum of entire file - let hash = compute_file_hash(&data); + // Hash check: a simple checksum of the entire file, streamed + let hash = compute_file_hash_streamed(path).map_err(cannot_read)?; checks.push((format!("BLAKE3 hash: {}", &hash[..16]), true)); } } @@ -958,12 +970,36 @@ fn verify_single_file(path: &Path, checks: &mut Vec<(String, bool)>) -> Result<( /// Compute a simple hash of file contents (using a basic checksum since we don't have blake3 dep). fn compute_file_hash(data: &[u8]) -> String { // FNV-1a 64-bit hash as lightweight integrity check - let mut hash: u64 = 0xcbf29ce484222325; + format!("{:016x}", fnv1a_update(FNV1A_OFFSET, data)) +} + +/// The FNV-1a 64-bit offset basis. +const FNV1A_OFFSET: u64 = 0xcbf29ce484222325; + +/// Fold `data` into an FNV-1a 64-bit state. The hash is a sequential byte fold, so feeding a +/// file chunk by chunk gives exactly the whole-buffer hash. +fn fnv1a_update(mut hash: u64, data: &[u8]) -> u64 { for &byte in data { hash ^= byte as u64; hash = hash.wrapping_mul(0x100000001b3); } - format!("{hash:016x}") + hash +} + +/// [`compute_file_hash`] of a file's bytes, read in 1 MiB chunks (#3761): the same hash, +/// without holding the file in memory. +fn compute_file_hash_streamed(path: &Path) -> std::io::Result { + use std::io::Read; + let mut file = std::fs::File::open(path)?; + let mut buf = vec![0u8; 1 << 20]; + let mut hash = FNV1A_OFFSET; + loop { + let n = file.read(&mut buf)?; + if n == 0 { + return Ok(format!("{hash:016x}")); + } + hash = fnv1a_update(hash, &buf[..n]); + } } // ── PPL-Benchmark Correlation (R-066) ─────────────────────────────────────── diff --git a/crates/apr-cli/src/commands/model_header.rs b/crates/apr-cli/src/commands/model_header.rs index cc2d96e07d..be2055358f 100644 --- a/crates/apr-cli/src/commands/model_header.rs +++ b/crates/apr-cli/src/commands/model_header.rs @@ -8,63 +8,24 @@ //! tensor data. Each needs a few bytes or the header, and these readers never //! touch more. -use std::io::Read; use std::path::Path; use aprender::format::gguf::reader::GgufReader; -use aprender::format::v2::{AprV2Header, HEADER_SIZE_V2}; +use aprender::format::prefix; -/// The first GGUF header prefix tried. Measured on 21 local GGUFs (2026-09-21): headers are -/// 5.7 MiB (Qwen2.5 / Qwen3) to 10.5 MiB (Qwen3.5, a 248,320-token vocabulary), on files up -/// to 17.3 GiB, so one read covers every model on the ladder. -const GGUF_HEADER_FIRST: usize = 16 << 20; - -/// The most a header read ever takes: about 24x the largest header measured. A file whose -/// header does not parse inside it is refused, never read whole. -const HEADER_CAP: usize = 256 << 20; - -/// At most the first `n` bytes of `path`. +/// At most the first `n` bytes of `path` (the shared policy, `aprender::format::prefix`). pub(crate) fn read_prefix(path: &Path, n: usize) -> std::io::Result> { - let mut buf = Vec::with_capacity(n.min(1 << 20)); - std::fs::File::open(path)? - .take(n as u64) - .read_to_end(&mut buf)?; - Ok(buf) + prefix::read_prefix(path, n) } -/// A GGUF header parsed from a growing prefix of the file: 16 MiB first, doubling, never -/// more than 256 MiB. +/// A GGUF header parsed from a growing prefix of the file, under the shared policy +/// (`prefix::HEADER_FIRST_READ` first, doubling, never more than `prefix::HEADER_READ_CAP`). /// /// The reader holds ONLY that prefix. Its metadata and tensor table are complete, and its /// tensor data is absent, so it answers header questions (architecture, the tensor /// `(name, type)` table, rope/context/eps) and nothing else. pub(crate) fn gguf_header(path: &Path) -> Result { - gguf_header_within(path, GGUF_HEADER_FIRST, HEADER_CAP) -} - -/// [`gguf_header`] with its two limits as parameters, so the case table can exercise the -/// growth and the cap on small files. -fn gguf_header_within(path: &Path, first: usize, cap: usize) -> Result { - let len = std::fs::metadata(path) - .map_err(|e| format!("cannot stat {}: {e}", path.display()))? - .len(); - let mut n = first.min(cap); - loop { - let prefix = - read_prefix(path, n).map_err(|e| format!("cannot read {}: {e}", path.display()))?; - let whole_file = prefix.len() as u64 >= len; - match GgufReader::from_bytes(prefix) { - Ok(reader) => return Ok(reader), - Err(e) if whole_file => return Err(format!("GGUF header parse failed: {e}")), - Err(e) if n >= cap => { - return Err(format!( - "no complete GGUF header in the first {cap} bytes of {} ({e}); refused rather than read whole", - path.display() - )) - }, - Err(_) => n = (n * 2).min(cap), - } - } + prefix::parse_growing_prefix(path, GgufReader::from_bytes) } /// The architecture and the tensor `(name, GGML type)` table of a GGUF, from its header. @@ -85,26 +46,119 @@ pub(crate) fn gguf_arch_and_tensors(path: &Path) -> Option<(String, Vec<(String, Some((arch, tensors)) } -/// The bytes of an APR v2 file before its tensor data: the header, the metadata and the -/// tensor index, which is everything `AprV2Reader::from_bytes` parses. The header's own -/// `data_offset` bounds it, and so does `HEADER_CAP`. +/// The complete GGUF header as bytes: everything before the tensor data, cut at the offset +/// the strict `GgufReader` parse measured. +/// +/// This is for parsers that are NOT safe on a guessed prefix. `LlamaTokenizer::from_gguf_bytes` +/// `break`s on a metadata key cut short and returns `Ok` with whatever it has read, so a prefix +/// that ends inside the header would silently lose the fields after the cut. Cut at +/// `data_offset`, the header is always whole. +pub(crate) fn gguf_header_bytes(path: &Path) -> Result, String> { + let header_len = gguf_header(path)?.data_offset; + read_prefix(path, header_len).map_err(|e| format!("cannot read {}: {e}", path.display())) +} + +/// The bytes of an APR v2 file before its tensor data (the shared policy's reader). pub(crate) fn apr_header_prefix(path: &Path) -> Result, String> { - let head = read_prefix(path, HEADER_SIZE_V2) - .map_err(|e| format!("cannot read {}: {e}", path.display()))?; - let header = - AprV2Header::from_bytes(&head).map_err(|e| format!("APR header parse failed: {e}"))?; - let n = usize::try_from(header.data_offset) - .ok() - .filter(|&n| n <= HEADER_CAP) - .ok_or_else(|| { - format!( - "APR data_offset {} is past the {} MiB header cap; refused rather than read whole", - header.data_offset, - HEADER_CAP >> 20 - ) - })?; - read_prefix(path, n.max(HEADER_SIZE_V2)) - .map_err(|e| format!("cannot read {}: {e}", path.display())) + prefix::apr_v2_header_prefix(path) +} + +/// The #3750 / #3761 case rows' shared probe. A row re-runs this test binary on ONE probe test +/// in a fresh process and reads the peak RSS that process reports: a child, because `cargo test` +/// shares one process between tests and its peak would be theirs. +#[cfg(all(test, target_os = "linux"))] +pub(crate) mod rss_probe { + use std::io::Write; + use std::path::Path; + + /// Every case row's sparse fixture is this long. + pub(crate) const FIXTURE_LEN: u64 = 2 << 30; + + /// Measured (x86-64 debug test binary, 2026-09-21), the probes' peak RSS was 44.4 MiB for + /// the qa header readers on a 2 GiB sparse GGUF, 59.5-60.3 MiB on a real 0.5B GGUF (5.7 MiB + /// header) and 57.4-59.0 MiB on the real Qwen3-30B-A3B GGUF (17.3 GiB). With a whole-file + /// `std::fs::read` put back into `gguf_arch_and_tensors` it was 2,136,776 KiB (2.04 GiB). + /// Each #3761 row records its own numbers beside its test. The bound sits about 4x above + /// the header paths and 8x below a regression. + pub(crate) const PEAK_RSS_BOUND_KB: u64 = 256 * 1024; + + /// `bytes`, then extended SPARSELY to `len`: a multi-GiB model file that costs no disk. + pub(crate) fn sparse(bytes: &[u8], len: u64, suffix: &str) -> tempfile::NamedTempFile { + let mut f = tempfile::NamedTempFile::with_suffix(suffix).expect("temp file"); + f.write_all(bytes).expect("write"); + f.as_file().set_len(len).expect("extend sparsely"); + f + } + + /// A GGUF (`llama`, one F32 tensor) whose header carries a four-token vocabulary. + pub(crate) fn gguf_with_vocab() -> Vec { + use aprender::format::gguf::{export_tensors_to_gguf, GgmlType, GgufTensor, GgufValue}; + let tensors = vec![GgufTensor { + name: "token_embd.weight".into(), + shape: vec![4, 8], + dtype: GgmlType::F32, + data: vec![0u8; 128], + }]; + let tokens = ["", "", "", "a"].map(String::from).to_vec(); + let metadata = vec![ + ( + "general.architecture".to_string(), + GgufValue::String("llama".into()), + ), + ( + "tokenizer.ggml.tokens".to_string(), + GgufValue::ArrayString(tokens), + ), + ]; + let mut bytes = Vec::new(); + export_tensors_to_gguf(&mut bytes, &tensors, &metadata).expect("write GGUF"); + bytes + } + + /// A SafeTensors file holding one F32 tensor. + pub(crate) fn safetensors() -> Vec { + let json = br#"{"w":{"dtype":"F32","shape":[1],"data_offsets":[0,4]}}"#; + let mut bytes = (json.len() as u64).to_le_bytes().to_vec(); + bytes.extend_from_slice(json); + bytes.extend_from_slice(&[0u8; 4]); + bytes + } + + /// Run the probe test `probe` in a child with `envs` set, and return the peak RSS (KiB) it + /// reported. Panics, with the child's output, when it reports none. + pub(crate) fn child_peak_kb(probe: &str, envs: &[(&str, &Path)]) -> u64 { + let mut cmd = + std::process::Command::new(std::env::current_exe().expect("this test binary")); + cmd.args(["--exact", probe, "--nocapture", "--test-threads=1"]); + for (k, v) in envs { + cmd.env(k, v); + } + let out = cmd.output().expect("run the probe"); + let text = String::from_utf8_lossy(&out.stdout); + let hwm = text + .lines() + .find_map(|l| l.split("PEAK_RSS_KB=").nth(1)) // libtest prints it after the test name + .and_then(|v| v.split_whitespace().next()?.parse::().ok()) + .unwrap_or_else(|| { + panic!( + "the probe {probe} reported no peak: {text}{}", + String::from_utf8_lossy(&out.stderr) + ) + }); + eprintln!("{probe}: peak RSS {hwm} KiB"); + hwm + } + + /// The probe side: print this process's peak RSS where [`child_peak_kb`] reads it. + pub(crate) fn report_peak() { + let status = std::fs::read_to_string("/proc/self/status").expect("/proc/self/status"); + let hwm = status + .lines() + .find_map(|l| l.strip_prefix("VmHWM:")) + .and_then(|v| v.trim().trim_end_matches("kB").trim().parse::().ok()) + .expect("VmHWM in /proc/self/status"); + println!("PEAK_RSS_KB={hwm}"); + } } #[cfg(test)] @@ -180,14 +234,16 @@ mod tests { #[test] fn a_header_larger_than_the_first_read_is_found_by_growing() { let f = gguf_fixture(0, 5_000); - let reader = gguf_header_within(f.path(), 1024, 1 << 20).expect("grows past 1 KiB"); + let reader = + prefix::parse_growing_prefix_within(f.path(), 1024, 1 << 20, GgufReader::from_bytes) + .expect("grows past 1 KiB"); assert_eq!(reader.architecture().as_deref(), Some("llama")); } #[test] fn a_header_past_the_cap_is_refused_never_read_whole() { let f = gguf_fixture(1 << 30, 5_000); - let e = gguf_header_within(f.path(), 1024, 4096) + let e = prefix::parse_growing_prefix_within(f.path(), 1024, 4096, GgufReader::from_bytes) .expect_err("a 4 KiB cap cannot hold a 5 KB header"); assert!(e.contains("refused rather than read whole"), "{e}"); } @@ -215,7 +271,7 @@ mod tests { f.write_all(&bytes).expect("write"); let prefix = apr_header_prefix(f.path()).expect("the header, metadata and index"); - let header = AprV2Header::from_bytes(&prefix).expect("header"); + let header = aprender::format::v2::AprV2Header::from_bytes(&prefix).expect("header"); assert_eq!( prefix.len() as u64, header.data_offset, @@ -232,45 +288,25 @@ mod tests { } /// The done_when row: on a sparse 2 GiB GGUF, the header readers' PEAK RSS stays small. - /// Measured in a CHILD process (this test binary, re-run on the probe below), because - /// `cargo test` shares one process between tests and its peak would be theirs. The bound - /// is measured, not guessed: see `PEAK_RSS_BOUND_KB`. + /// The bound is measured, not guessed: see `rss_probe::PEAK_RSS_BOUND_KB`. #[cfg(target_os = "linux")] #[test] fn header_reads_of_a_2_gib_gguf_keep_peak_rss_small() { - let f = gguf_fixture(2 << 30, 0); - let out = std::process::Command::new(std::env::current_exe().expect("this test binary")) - .args([ - "--exact", - "commands::model_header::tests::peak_rss_probe", - "--nocapture", - "--test-threads=1", - ]) - .env(PROBE_ENV, f.path()) - .output() - .expect("run the probe"); - let text = String::from_utf8_lossy(&out.stdout); - let hwm = text - .lines() - .find_map(|l| l.split("PEAK_RSS_KB=").nth(1)) // libtest prints it after the test name - .and_then(|v| v.split_whitespace().next()?.parse::().ok()) - .unwrap_or_else(|| panic!("the probe reported no peak: {text}{}", String::from_utf8_lossy(&out.stderr))); + let f = gguf_fixture(rss_probe::FIXTURE_LEN, 0); + let hwm = rss_probe::child_peak_kb( + "commands::model_header::tests::peak_rss_probe", + &[(PROBE_ENV, f.path())], + ); assert!( - hwm < PEAK_RSS_BOUND_KB, - "peak RSS {hwm} KiB reading the header of a 2 GiB GGUF (bound {PEAK_RSS_BOUND_KB} KiB): \ - a whole-file read is back" + hwm < rss_probe::PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB reading the header of a 2 GiB GGUF (bound {} KiB): \ + a whole-file read is back", + rss_probe::PEAK_RSS_BOUND_KB ); } const PROBE_ENV: &str = "APR_3750_PEAK_RSS_PROBE"; - /// Measured on this commit (x86-64 debug test binary, 2026-09-21), the probe's peak RSS was: - /// 44.4 MiB on this 2 GiB sparse fixture; 59.5-60.3 MiB on a real 0.5B GGUF (5.7 MiB header); - /// 57.4-59.0 MiB on the real Qwen3-30B-A3B GGUF (17.3 GiB). With a whole-file `std::fs::read` - /// put back into `gguf_arch_and_tensors`, it was 2,136,776 KiB (2.04 GiB) on this fixture. The - /// bound sits about 4x above the header path and 8x below the regression. - const PEAK_RSS_BOUND_KB: u64 = 256 * 1024; - /// Not a test on its own: with the probe variable unset it does nothing. The test above /// re-runs this binary on it, in a fresh process, to read that process's peak RSS. #[cfg(target_os = "linux")] @@ -285,14 +321,12 @@ mod tests { "the probe's fixture has a header" ); assert!(gguf_header(&path).is_ok()); + // #3761: the whole header as bytes, for the parsers that are not prefix-safe + let header = gguf_header_bytes(&path).expect("the header bytes"); let _ = super::super::qa_capability::cpu_only_architecture(&path); let _ = super::super::qa_capability::hybrid_loader_architecture(&path); - let status = std::fs::read_to_string("/proc/self/status").expect("/proc/self/status"); - let hwm = status - .lines() - .find_map(|l| l.strip_prefix("VmHWM:")) - .and_then(|v| v.trim().trim_end_matches("kB").trim().parse::().ok()) - .expect("VmHWM in /proc/self/status"); - println!("PEAK_RSS_KB={hwm}"); + // Reported before the length check, so a whole-file read fails the row on its peak + rss_probe::report_peak(); + assert!(header.len() < 1 << 20, "{} header bytes", header.len()); } } diff --git a/crates/apr-format/src/lib.rs b/crates/apr-format/src/lib.rs index 0a366ce703..0955458e1e 100644 --- a/crates/apr-format/src/lib.rs +++ b/crates/apr-format/src/lib.rs @@ -49,6 +49,7 @@ pub mod error; pub mod f16; pub mod falsifiers; pub mod model_card; +pub mod prefix; pub mod types; pub mod v2; pub mod validate; diff --git a/crates/apr-format/src/prefix.rs b/crates/apr-format/src/prefix.rs new file mode 100644 index 0000000000..16d2f9323b --- /dev/null +++ b/crates/apr-format/src/prefix.rs @@ -0,0 +1,253 @@ +//! Bounded reads of a model file's head: the ONE policy for "read the header, never the +//! tensor data" (#3750, #3761). +//! +//! Reading a whole model to learn its format, architecture or metadata costs its full size in +//! RSS: 17.3 GiB for Qwen3-30B-A3B, and on GB10's unified memory that competes with the GPU +//! (the 2026-09-21 OOM killed 18 CI containers). Every reader that needs only the head of a +//! file goes through here, so the limits live in one place. + +use std::fmt::Display; +use std::io::Read; +use std::path::Path; + +use crate::v2::{AprV2Header, HEADER_SIZE_V2}; + +/// The first header read. Measured on 21 local GGUFs (2026-09-21): headers are 5.7 MiB +/// (Qwen2.5 / Qwen3) to 10.5 MiB (Qwen3.5, a 248,320-token vocabulary), on files up to +/// 17.3 GiB, so one read covers every model on the ladder. +pub const HEADER_FIRST_READ: usize = 16 << 20; + +/// The most a header read ever takes: about 24x the largest header measured. A header that +/// does not fit inside it is refused, and the file is never read whole. +pub const HEADER_READ_CAP: usize = 256 << 20; + +/// At most the first `n` bytes of `path`. +pub fn read_prefix(path: &Path, n: usize) -> std::io::Result> { + let mut buf = Vec::with_capacity(n.min(1 << 20)); + std::fs::File::open(path)? + .take(n as u64) + .read_to_end(&mut buf)?; + Ok(buf) +} + +/// Parse a header from a growing prefix of `path`: [`HEADER_FIRST_READ`] first, doubling, +/// never more than [`HEADER_READ_CAP`]. `parse` must fail on a truncated header, as every +/// bounds-checked header parser does. A parse that fails once the whole file has been read +/// is that parser's error. A parse that still fails at the cap is refused by name. +pub fn parse_growing_prefix( + path: &Path, + parse: impl FnMut(Vec) -> Result, +) -> Result { + parse_growing_prefix_within(path, HEADER_FIRST_READ, HEADER_READ_CAP, parse) +} + +/// [`parse_growing_prefix`] with its two limits as parameters (the case table uses small ones). +pub fn parse_growing_prefix_within( + path: &Path, + first: usize, + cap: usize, + mut parse: impl FnMut(Vec) -> Result, +) -> Result { + let len = std::fs::metadata(path) + .map_err(|e| format!("cannot stat {}: {e}", path.display()))? + .len(); + let mut n = first.min(cap); + loop { + let prefix = + read_prefix(path, n).map_err(|e| format!("cannot read {}: {e}", path.display()))?; + let whole_file = prefix.len() as u64 >= len; + match parse(prefix) { + Ok(v) => return Ok(v), + Err(e) if whole_file => return Err(format!("header parse failed: {e}")), + Err(e) if n >= cap => { + return Err(format!( + "no complete header in the first {cap} bytes of {} ({e}); refused rather than read whole", + path.display() + )) + } + Err(_) => n = (n * 2).min(cap), + } + } +} + +/// The bytes of an APR v2 file before its tensor data: the header, the metadata and the +/// tensor index, which is everything `AprV2Reader::from_bytes` parses. The header's own +/// `data_offset` bounds it, and so does [`HEADER_READ_CAP`]. +pub fn apr_v2_header_prefix(path: &Path) -> Result, String> { + let head = read_prefix(path, HEADER_SIZE_V2) + .map_err(|e| format!("cannot read {}: {e}", path.display()))?; + let header = + AprV2Header::from_bytes(&head).map_err(|e| format!("APR header parse failed: {e}"))?; + let n = usize::try_from(header.data_offset) + .ok() + .filter(|&n| n <= HEADER_READ_CAP) + .ok_or_else(|| { + format!( + "APR data_offset {} is past the {} MiB header cap; refused rather than read whole", + header.data_offset, + HEADER_READ_CAP >> 20 + ) + })?; + read_prefix(path, n.max(HEADER_SIZE_V2)) + .map_err(|e| format!("cannot read {}: {e}", path.display())) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + + /// A temp file removed on drop (the leaf takes no test dependency for this). + struct TempFile(std::path::PathBuf); + impl TempFile { + fn path(&self) -> &Path { + &self.0 + } + } + impl Drop for TempFile { + fn drop(&mut self) { + let _ = std::fs::remove_file(&self.0); + } + } + + fn file_with(bytes: &[u8]) -> TempFile { + use std::sync::atomic::{AtomicUsize, Ordering}; + static N: AtomicUsize = AtomicUsize::new(0); + let path = std::env::temp_dir().join(format!( + "apr-format-prefix-{}-{}", + std::process::id(), + N.fetch_add(1, Ordering::Relaxed) + )); + std::fs::File::create(&path) + .and_then(|mut f| f.write_all(bytes)) + .expect("write the temp file"); + TempFile(path) + } + + /// A toy header: its first byte says how long the whole header is. It takes the `Vec` because + /// `parse_growing_prefix` hands its parser ownership of each prefix. + #[allow(clippy::needless_pass_by_value)] + fn toy_parse(prefix: Vec) -> Result { + let need = *prefix.first().ok_or("empty")? as usize; + if prefix.len() >= need { + Ok(need) + } else { + Err(format!("truncated: have {}, need {need}", prefix.len())) + } + } + + /// A file of `bytes`, then extended SPARSELY to `len`: a multi-GiB model that costs no disk. + fn sparse_file_with(bytes: &[u8], len: u64) -> TempFile { + let f = file_with(bytes); + std::fs::OpenOptions::new() + .write(true) + .open(f.path()) + .and_then(|h| h.set_len(len)) + .expect("extend sparsely"); + f + } + + fn apr_v2_bytes() -> Vec { + use crate::v2::{AprV2Metadata, AprV2Writer, TensorDType}; + let mut writer = AprV2Writer::new(AprV2Metadata::new("test")); + writer.add_tensor("w", TensorDType::F32, vec![4, 4], vec![0u8; 64]); + let mut bytes = Vec::new(); + writer.write_to(&mut bytes).expect("write APR"); + bytes + } + + const PROBE_APR: &str = "APR_3761_PROBE_APR"; + + /// Measured (x86-64 debug test binary, 2026-09-21): this probe's peak RSS was 4,608 KiB on three runs. + /// With a whole-file `std::fs::read` put back into `apr_v2_header_prefix` it was 2,094,336 KiB + /// on the 2 GiB fixture. The bound sits far between the two. (The SafeTensors header reader + /// lives in aprender-core, beside its format, and has its row there.) + const PEAK_RSS_BOUND_KB: u64 = 256 * 1024; + + /// The case row the #3761 done_when names: on a sparse 2 GiB APR v2 file, the header reader + /// keeps PEAK RSS small. Measured in a CHILD process (this test binary, + /// re-run on the probe below), because `cargo test` shares one process between tests. + #[cfg(target_os = "linux")] + #[test] + fn header_readers_keep_peak_rss_small_on_2_gib_files() { + let apr = sparse_file_with(&apr_v2_bytes(), 2 << 30); + let out = std::process::Command::new(std::env::current_exe().expect("this test binary")) + .args([ + "--exact", + "prefix::tests::peak_rss_probe", + "--nocapture", + "--test-threads=1", + ]) + .env(PROBE_APR, apr.path()) + .output() + .expect("run the probe"); + let text = String::from_utf8_lossy(&out.stdout); + let hwm = text + .lines() + .find_map(|l| l.split("PEAK_RSS_KB=").nth(1)) + .and_then(|v| v.split_whitespace().next()?.parse::().ok()) + .unwrap_or_else(|| { + panic!( + "the probe reported no peak: {text}{}", + String::from_utf8_lossy(&out.stderr) + ) + }); + eprintln!("prefix::tests::peak_rss_probe: peak RSS {hwm} KiB"); + assert!( + hwm < PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB reading the header of a 2 GiB APR file (bound {PEAK_RSS_BOUND_KB} KiB): a whole-file read is back" + ); + } + + /// Not a test on its own: with the probe variable unset it does nothing. + #[cfg(target_os = "linux")] + #[test] + fn peak_rss_probe() { + let Some(apr) = std::env::var_os(PROBE_APR) else { + return; + }; + let apr = apr_v2_header_prefix(Path::new(&apr)).expect("the APR header prefix"); + assert!(crate::v2::AprV2Header::from_bytes(&apr).is_ok()); + let status = std::fs::read_to_string("/proc/self/status").expect("/proc/self/status"); + let hwm = status + .lines() + .find_map(|l| l.strip_prefix("VmHWM:")) + .and_then(|v| v.trim().trim_end_matches("kB").trim().parse::().ok()) + .expect("VmHWM in /proc/self/status"); + println!("PEAK_RSS_KB={hwm}"); + } + + #[test] + fn read_prefix_reads_at_most_n_bytes() { + let f = file_with(b"0123456789"); + assert_eq!(read_prefix(f.path(), 4).expect("read"), b"0123"); + assert_eq!(read_prefix(f.path(), 100).expect("read"), b"0123456789"); + } + + #[test] + fn a_header_larger_than_the_first_read_is_found_by_growing() { + let mut bytes = vec![200u8]; + bytes.resize(4096, 0); + let f = file_with(&bytes); + assert_eq!( + parse_growing_prefix_within(f.path(), 16, 1024, toy_parse), + Ok(200) + ); + } + + #[test] + fn a_header_past_the_cap_is_refused_never_read_whole() { + let mut bytes = vec![200u8]; + bytes.resize(4096, 0); + let f = file_with(&bytes); + let e = parse_growing_prefix_within(f.path(), 16, 64, toy_parse).expect_err("200 > 64"); + assert!(e.contains("refused rather than read whole"), "{e}"); + } + + #[test] + fn a_parse_that_fails_on_the_whole_file_is_the_parsers_error() { + let f = file_with(&[100u8, 1, 2]); + let e = parse_growing_prefix_within(f.path(), 16, 1024, toy_parse).expect_err("3 < 100"); + assert!(e.starts_with("header parse failed: truncated"), "{e}"); + } +} diff --git a/crates/aprender-core/src/format/converter/apr_export_fn.rs b/crates/aprender-core/src/format/converter/apr_export_fn.rs index a9a389b9e5..18da4716a4 100644 --- a/crates/aprender-core/src/format/converter/apr_export_fn.rs +++ b/crates/aprender-core/src/format/converter/apr_export_fn.rs @@ -130,7 +130,8 @@ fn enforce_export_completeness( /// Detect architecture from APR metadata for completeness checking. fn detect_apr_architecture_for_completeness(apr_path: &Path) -> Option<&'static str> { - let data = fs::read(apr_path).ok()?; + // #3761: metadata only; the header prefix, never the tensor data + let data = crate::format::prefix::apr_v2_header_prefix(apr_path).ok()?; let reader = crate::format::v2::AprV2Reader::from_bytes(&data).ok()?; let metadata = reader.metadata(); let arch = metadata.architecture.as_deref().or_else(|| { diff --git a/crates/aprender-core/src/format/converter/export.rs b/crates/aprender-core/src/format/converter/export.rs index 7940462037..04dd9a1529 100644 --- a/crates/aprender-core/src/format/converter/export.rs +++ b/crates/aprender-core/src/format/converter/export.rs @@ -469,3 +469,7 @@ fn shape_to_gguf(shape: &[usize]) -> Vec { include!("export_include.rs"); include!("fusion.rs"); include!("apr_export_fn.rs"); + +#[cfg(all(test, target_os = "linux"))] +#[path = "export_rss_tests.rs"] +mod rss_tests; diff --git a/crates/aprender-core/src/format/converter/export_rss_tests.rs b/crates/aprender-core/src/format/converter/export_rss_tests.rs new file mode 100644 index 0000000000..602e4751ae --- /dev/null +++ b/crates/aprender-core/src/format/converter/export_rss_tests.rs @@ -0,0 +1,51 @@ +//! #3761 case row: the converter's APR readers (`read_apr_metadata`, `extract_user_metadata`, +//! `detect_apr_quantization`, `detect_apr_architecture_for_completeness`, +//! `extract_apr_tokenizer_hint`) read a 2 GiB APR v2 file's header, metadata and tensor index, +//! and never its tensor data. Each read the whole model. +//! +//! Measured (x86-64 debug, 2026-09-21): 16,200-17,648 KiB over eight runs. A whole-file read put +//! back into any one of the five: 2,110,308-4,208,572 KiB. + +use super::*; +use crate::format::prefix_rss_tests::{child_peak_kb, report_peak, sparse, PEAK_RSS_BOUND_KB}; + +const PROBE: &str = "APR_3761_CONVERTER_PROBE"; + +#[test] +fn converter_apr_readers_keep_peak_rss_small_on_a_2_gib_model() { + let apr = sparse(&crate::format::prefix_rss_tests::apr_bytes(), ".apr"); + let hwm = child_peak_kb( + "format::converter::export::rss_tests::peak_rss_probe", + &[(PROBE, apr.path())], + ); + assert!( + hwm < PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB reading a 2 GiB APR's metadata (bound {PEAK_RSS_BOUND_KB} KiB): \ + a whole-file read is back" + ); +} + +/// Not a test on its own: with the probe variable unset it does nothing. +#[test] +fn peak_rss_probe() { + let Some(path) = std::env::var_os(PROBE) else { + return; + }; + let path = Path::new(&path); + let metadata = read_apr_metadata(path); + let user = extract_user_metadata(path); + let quantization = detect_apr_quantization(path); + let arch = detect_apr_architecture_for_completeness(path); + let hint = extract_apr_tokenizer_hint(path); + // Reported before the result checks, so a whole-file read fails the row on its peak + report_peak(); + assert_eq!( + metadata.and_then(|m| m.architecture).as_deref(), + Some("llama") + ); + assert_eq!(user.get("origin").map(String::as_str), Some("fixture")); + assert!(quantization.is_none(), "an F32 model is not quantized"); + assert_eq!(arch, Some("llama")); + // The v2 writer zero-pads its metadata: no terminator, so no hint (see the reader) + assert!(hint.is_none()); +} diff --git a/crates/aprender-core/src/format/converter/export_tests_infer_attn.rs b/crates/aprender-core/src/format/converter/export_tests_infer_attn.rs index 871e1bf2df..16fb5b2c16 100644 --- a/crates/aprender-core/src/format/converter/export_tests_infer_attn.rs +++ b/crates/aprender-core/src/format/converter/export_tests_infer_attn.rs @@ -378,6 +378,30 @@ fn test_infer_tokenizer_json_apr_without_tokenizer() { let _ = fs::remove_file(&path); } +// ======================================================================== +// infer_tokenizer_json: the legacy layout with a tokenizer (#3761) +// ======================================================================== + +/// The positive control for the hint reader's non-v2 path, which #3761 moved from a whole-file +/// read to a growing prefix: a 44-byte head, then metadata naming a tokenizer and ending in the +/// terminator, then data. The hint is still found. +#[test] +fn test_infer_tokenizer_json_legacy_apr_with_tokenizer() { + let dir = std::env::temp_dir().join("apr_test_legacy_tokenizer_3761"); + let _ = fs::create_dir_all(&dir); + let path = dir.join("legacy_tok.apr"); + let mut data = vec![0u8; 44]; + data.extend_from_slice(br#"{"tokenizer": {"type": "BPE"}"#); + data.extend_from_slice(b"}\n\n\n"); + data.extend_from_slice(&[0u8; 4096]); + fs::write(&path, &data).expect("write failed"); + + let result = infer_tokenizer_json(&path); + assert!(result.contains("BPE"), "{result:?}"); + + let _ = fs::remove_file(&path); +} + // ======================================================================== // GH-253-4: ValidatedGgufMetadata tests // ======================================================================== diff --git a/crates/aprender-core/src/format/converter/tensor.rs b/crates/aprender-core/src/format/converter/tensor.rs index 68e5fdc4b5..9b887790f4 100644 --- a/crates/aprender-core/src/format/converter/tensor.rs +++ b/crates/aprender-core/src/format/converter/tensor.rs @@ -198,17 +198,29 @@ fn infer_tokenizer_json(input_path: &Path) -> String { } /// Try to extract tokenizer hint from APR metadata section. +/// +/// #3761: never the tensor data. An APR v2 file's metadata ends before its `data_offset`, so +/// its header prefix bounds the scan. (The v2 writer zero-pads the metadata, so the terminator +/// this scan looks for is not there, and the answer is `None`, as it was when the scan read +/// the whole model.) Any other layout is scanned over a growing prefix under the shared cap, +/// the prefix growing only until the terminator is found. fn extract_apr_tokenizer_hint(input_path: &Path) -> Option { - let data = fs::read(input_path).ok()?; - if data.len() <= 44 { - return None; - } - let metadata_start = 44; - let metadata_end = data[metadata_start..] - .windows(4) - .position(|w| w == b"}\n\n\n" || w == b"}\r\n\r") - .map(|p| metadata_start + p + 1)?; - let metadata_str = std::str::from_utf8(&data[metadata_start..metadata_end]).ok()?; + const METADATA_START: usize = 44; + let scan = |data: Vec| -> std::result::Result { + let end = data + .get(METADATA_START..) + .and_then(|tail| { + tail.windows(4) + .position(|w| w == b"}\n\n\n" || w == b"}\r\n\r") + }) + .map(|p| METADATA_START + p + 1) + .ok_or("no metadata terminator in this prefix")?; + String::from_utf8(data[METADATA_START..end].to_vec()).map_err(|_| "metadata is not UTF-8") + }; + let metadata_str = match crate::format::prefix::apr_v2_header_prefix(input_path) { + Ok(v2_head) => scan(v2_head).ok()?, + Err(_) => crate::format::prefix::parse_growing_prefix(input_path, scan).ok()?, + }; if metadata_str.contains("\"tokenizer\"") || metadata_str.contains("\"vocabulary\"") { Some(r#"{"version": "1.0", "model": {"type": "BPE"}}"#.to_string()) } else { @@ -223,7 +235,8 @@ fn read_apr_metadata(apr_path: &Path) -> Option UserMetadata { - let data = match fs::read(apr_path) { - Ok(d) => d, - Err(_) => return UserMetadata::new(), - }; - - // Real APR v2 header (header_impl.rs::to_bytes, 64 bytes): magic[0..4], version[4..6], - // flags[6..8], tensor_count u32 [8..12], metadata_offset u64 [12..20], metadata_size u32 - // [20..24]; the metadata JSON begins at `metadata_offset` (= HEADER_SIZE_V2 = 64). The prior - // code read an 8-byte "metadata_len" at byte 8 (= tensor_count | metadata_offset<<32 ≈ 2.7e11) - // and the JSON at byte 16, so the bounds guard ALWAYS failed and this returned empty — silently - // dropping the user's SafeTensors __metadata__ on every `apr export`. - if data.len() < 24 { + let Some(parsed) = read_apr_metadata_json(apr_path) else { return UserMetadata::new(); - } - let metadata_offset = u64::from_le_bytes(data[12..20].try_into().unwrap_or([0u8; 8])) as usize; - let metadata_size = u32::from_le_bytes(data[20..24].try_into().unwrap_or([0u8; 4])) as usize; - let end = match metadata_offset.checked_add(metadata_size) { - Some(e) if e <= data.len() => e, - _ => return UserMetadata::new(), - }; - - let metadata_json = match std::str::from_utf8(&data[metadata_offset..end]) { - Ok(s) => s, - Err(_) => return UserMetadata::new(), - }; - - let parsed: serde_json::Value = match serde_json::from_str(metadata_json) { - Ok(v) => v, - Err(_) => return UserMetadata::new(), }; // `custom` is #[serde(flatten)] in AprV2Metadata, so "source_metadata" is at the TOP level @@ -428,6 +414,26 @@ fn extract_user_metadata(apr_path: &Path) -> UserMetadata { UserMetadata::new() } +/// The APR v2 metadata section, parsed (#3761): the 24-byte header, then only up to the end of +/// the metadata section it names, under the shared cap. Never the tensor data. `None` for a file +/// too short, a section past the cap, or text that is not UTF-8 JSON. +/// +/// Real APR v2 header (header_impl.rs::to_bytes, 64 bytes): magic[0..4], version[4..6], +/// flags[6..8], tensor_count u32 [8..12], metadata_offset u64 [12..20], metadata_size u32 +/// [20..24]; the metadata JSON begins at `metadata_offset` (= HEADER_SIZE_V2 = 64). The prior +/// code read an 8-byte "metadata_len" at byte 8 (= tensor_count | metadata_offset<<32 ≈ 2.7e11) +/// and the JSON at byte 16, so the bounds guard ALWAYS failed and this returned empty — silently +/// dropping the user's SafeTensors __metadata__ on every `apr export`. +fn read_apr_metadata_json(apr_path: &Path) -> Option { + use crate::format::prefix::{read_prefix, HEADER_READ_CAP}; + let head = read_prefix(apr_path, 24).ok()?; + let offset = usize::try_from(u64::from_le_bytes(head.get(12..20)?.try_into().ok()?)).ok()?; + let size = u32::from_le_bytes(head.get(20..24)?.try_into().ok()?) as usize; + let end = offset.checked_add(size).filter(|&e| e <= HEADER_READ_CAP)?; + let data = read_prefix(apr_path, end).ok().filter(|d| d.len() >= end)?; + serde_json::from_str(std::str::from_utf8(data.get(offset..end)?).ok()?).ok() +} + /// Detect predominant quantization type from an APR file (PMAT-252). /// /// Reads the tensor index and checks the dtype of 2D weight tensors. @@ -436,7 +442,8 @@ fn extract_user_metadata(apr_path: &Path) -> UserMetadata { pub(crate) fn detect_apr_quantization(apr_path: &Path) -> Option { use crate::format::v2::{AprV2Reader, TensorDType}; - let data = fs::read(apr_path).ok()?; + // #3761: dtypes live in the tensor index; the header prefix, never the tensor data + let data = crate::format::prefix::apr_v2_header_prefix(apr_path).ok()?; let reader = AprV2Reader::from_bytes(&data).ok()?; // Count dtypes across 2D weight tensors (skip 1D biases/norms) diff --git a/crates/aprender-core/src/format/gguf/api.rs b/crates/aprender-core/src/format/gguf/api.rs index eba45f1764..fee7270218 100644 --- a/crates/aprender-core/src/format/gguf/api.rs +++ b/crates/aprender-core/src/format/gguf/api.rs @@ -340,12 +340,7 @@ pub fn load_gguf_raw>(path: P) -> Result { // Propagate raw GGUF KV metadata to downstream consumers (inspect/rosetta) so they // can display authentic on-disk keys instead of fabricated ML-shorthand names. // #3733: EVERY header key, including those outside the reader's parse allowlist. - let raw_metadata: BTreeMap = reader - .metadata - .iter() - .chain(&reader.display_only_metadata) - .map(|(k, v)| (k.clone(), gguf_value_display(v))) - .collect(); + let raw_metadata = gguf_raw_metadata(&reader); Ok(GgufRawLoadResult { tensors, @@ -355,6 +350,18 @@ pub fn load_gguf_raw>(path: P) -> Result { }) } +/// Every header key of `reader` with its display string: parsed keys and the +/// display-only ones (#3733). Shared by [`load_gguf_raw`] and header-only `apr inspect` +/// (#4520 step 2), so both print the same metadata. +pub fn gguf_raw_metadata(reader: &GgufReader) -> BTreeMap { + reader + .metadata + .iter() + .chain(&reader.display_only_metadata) + .map(|(k, v)| (k.clone(), gguf_value_display(v))) + .collect() +} + /// Format a `GgufValue` as a human-readable display string. /// Arrays are summarized as `[len=N]` to keep display output bounded. /// Contract: apr-inspect-metadata-propagation-v1 F-INSPECT-META-001 (paiml/aprender#622). @@ -383,3 +390,7 @@ fn gguf_value_display(v: &crate::format::gguf::types::GgufValue) -> String { #[cfg(test)] #[path = "api_tests.rs"] mod tests; + +#[cfg(test)] +#[path = "header_from_file_tests.rs"] +mod header_from_file_tests; diff --git a/crates/aprender-core/src/format/gguf/header_from_file_tests.rs b/crates/aprender-core/src/format/gguf/header_from_file_tests.rs new file mode 100644 index 0000000000..91c2bcc211 --- /dev/null +++ b/crates/aprender-core/src/format/gguf/header_from_file_tests.rs @@ -0,0 +1,133 @@ +//! #4520 step 2: `GgufReader::header_from_file` reads the header, never the tensor data, +//! and sizes and refuses tensors exactly as the whole-file reader does. + +use super::*; +use crate::format::gguf::{export_tensors_to_gguf, GgmlType, GgufTensor, GgufValue}; +use std::io::Write; + +fn gguf(metadata: &[(String, GgufValue)]) -> Vec { + let tensors = vec![ + GgufTensor { + name: "b.weight".into(), + shape: vec![4, 8], + dtype: GgmlType::F32, + data: vec![1u8; 128], + }, + GgufTensor { + name: "a.weight".into(), + shape: vec![2, 2], + dtype: GgmlType::F32, + data: vec![2u8; 16], + }, + ]; + let mut bytes = Vec::new(); + export_tensors_to_gguf(&mut bytes, &tensors, metadata).expect("write GGUF"); + bytes +} + +fn arch() -> Vec<(String, GgufValue)> { + vec![( + "general.architecture".to_string(), + GgufValue::String("llama".into()), + )] +} + +fn file(bytes: &[u8]) -> tempfile::NamedTempFile { + let mut f = tempfile::NamedTempFile::with_suffix(".gguf").expect("temp file"); + f.write_all(bytes).expect("write"); + f +} + +#[test] +fn header_matches_the_whole_file_reader() { + let f = file(&gguf(&arch())); + let whole = load_gguf_raw(f.path()).expect("whole-file load"); + let (reader, len) = GgufReader::header_from_file(f.path()).expect("header"); + assert_eq!(len, std::fs::metadata(f.path()).expect("stat").len()); + assert_eq!(gguf_raw_metadata(&reader), whole.raw_metadata); + assert_eq!(reader.architecture(), whole.model_config.architecture); + let extents = reader.tensor_extents(len).expect("extents"); + let from_header: Vec<_> = extents + .iter() + .map(|(n, (shape, dtype, size))| (n.clone(), shape.clone(), *dtype, *size)) + .collect(); + let from_data: Vec<_> = whole + .tensors + .iter() + .map(|(n, t)| (n.clone(), t.shape.clone(), t.dtype, t.data.len())) + .collect(); + assert_eq!(from_header, from_data); +} + +fn big_vocab_gguf() -> Vec { + let mut metadata = arch(); + let tokens: Vec = (0..700_000).map(|i| format!("token-{i:08}")).collect(); + metadata.push(( + "tokenizer.ggml.tokens".to_string(), + GgufValue::ArrayString(tokens), + )); + gguf(&metadata) +} + +/// A header past the first prefix (a big vocabulary does this) still parses: the prefix +/// doubles instead of reporting the file as malformed. +#[test] +fn a_header_longer_than_the_first_prefix_still_parses() { + let bytes = big_vocab_gguf(); + assert!( + bytes.len() > 1 << 20, + "the fixture must outgrow the first prefix" + ); + let f = file(&bytes); + let (reader, len) = + GgufReader::header_from_file_within(f.path(), 1 << 20, 64 << 20).expect("header"); + assert_eq!(len, bytes.len() as u64); + assert_eq!(reader.tensor_extents(len).expect("extents").len(), 2); + assert_eq!( + gguf_raw_metadata(&reader) + .get("tokenizer.ggml.tokens") + .map(String::as_str), + Some("[len=700000]") + ); +} + +/// A header still unparsed at the cap is refused by name, never read to EOF: without the +/// cap a corrupt 17 GB GGUF buffered all 17 GB before its parse error (quorum finding on +/// PMAT-3761, lane 2). +#[test] +fn a_header_past_the_cap_is_refused_not_read_whole() { + let bytes = big_vocab_gguf(); + let f = file(&bytes); + let err = GgufReader::header_from_file_within(f.path(), 64 << 10, 1 << 20) + .expect_err("the header does not fit in 1 MiB"); + assert!( + err.to_string().contains("refused rather than read whole"), + "{err}" + ); +} + +/// A file cut inside its tensor data is refused as the whole-file reader refuses it. +#[test] +fn truncated_tensor_data_is_refused_from_the_header() { + let bytes = gguf(&arch()); + // 64 bytes: past the end-of-file alignment padding, into the last tensor. + let f = file(&bytes[..bytes.len() - 64]); + let whole = load_gguf_raw(f.path()).expect_err("whole-file load refuses it"); + let (reader, len) = GgufReader::header_from_file(f.path()).expect("the header is whole"); + let header = reader + .tensor_extents(len) + .expect_err("header-only refuses it"); + assert_eq!(header.to_string(), whole.to_string()); + assert!( + header.to_string().contains("data exceeds file size"), + "{header}" + ); +} + +/// Not a GGUF: the parse error is final once the whole (small) file has been read. +#[test] +fn a_non_gguf_file_is_refused() { + let f = file(&[0x42u8; 64]); + let err = GgufReader::header_from_file(f.path()).expect_err("not a GGUF"); + assert!(err.to_string().contains("Invalid GGUF magic"), "{err}"); +} diff --git a/crates/aprender-core/src/format/gguf/reader_parsing.rs b/crates/aprender-core/src/format/gguf/reader_parsing.rs index 5b8d5ea65a..2f11146e40 100644 --- a/crates/aprender-core/src/format/gguf/reader_parsing.rs +++ b/crates/aprender-core/src/format/gguf/reader_parsing.rs @@ -7,6 +7,48 @@ impl GgufReader { Self::from_bytes(data) } + /// Parse ONLY the header (metadata + tensor infos) of a GGUF file, never its tensor data. + /// + /// #4520 step 2 / #3761: `apr inspect --json` on Qwen3.5-27B-Q4_K_M read all 16.7 GB + /// and peaked at 32.8 GB RSS (`from_file`, then a copy of every tensor) to report + /// numbers the header holds. The header is read as a prefix that doubles until it + /// parses (16 MiB first, 256 MiB cap, the #3750 policy). A small file that does not + /// parse is that parse error; a header still unparsed at the cap is refused rather + /// than read whole. The reader's `data` is the prefix, so tensor bytes are NOT available from + /// it: size tensors with [`Self::tensor_extents`] against the returned file length. + pub fn header_from_file>(path: P) -> Result<(Self, u64)> { + Self::header_from_file_within( + path, + crate::format::prefix::HEADER_FIRST_READ, + crate::format::prefix::HEADER_READ_CAP, + ) + } + + /// [`Self::header_from_file`] with the first read and the cap as parameters (the case + /// table uses small ones). The prefix grows through the ONE bounded-prefix policy + /// (`parse_growing_prefix_within`), so a header that still does not parse at `cap` is + /// refused by name, never read to EOF: a corrupt 17 GB GGUF costs `cap`, not 17 GB. + pub fn header_from_file_within>( + path: P, + first: usize, + cap: usize, + ) -> Result<(Self, u64)> { + let path = path.as_ref(); + let file_len = std::fs::metadata(path).map_err(AprenderError::Io)?.len(); + // A FormatError contributes its message, not its Display: the prefix policy + // wraps it in its own FormatError, and "Invalid model format: … Invalid model + // format: …" is the core prefix leaking under the CLI's (#3661). + let parse = |bytes: Vec| { + Self::from_bytes(bytes).map_err(|e| match e { + AprenderError::FormatError { message } => message, + other => other.to_string(), + }) + }; + let reader = crate::format::prefix::parse_growing_prefix_within(path, first, cap, parse) + .map_err(|message| AprenderError::FormatError { message })?; + Ok((reader, file_len)) + } + /// Load a GGUF file preserving ALL metadata keys (no architecture /// whitelist). /// diff --git a/crates/aprender-core/src/format/gguf/shape.rs b/crates/aprender-core/src/format/gguf/shape.rs index d63c593d1f..f857886189 100644 --- a/crates/aprender-core/src/format/gguf/shape.rs +++ b/crates/aprender-core/src/format/gguf/shape.rs @@ -214,7 +214,37 @@ impl GgufReader { .ok_or_else(|| AprenderError::FormatError { message: format!("Tensor '{name}' not found in GGUF"), })?; + let (tensor_start, byte_size, shape) = self.tensor_extent(meta, self.data.len())?; + let bytes = self.data[tensor_start..tensor_start + byte_size].to_vec(); + Ok((bytes, shape, meta.dtype)) + } + + /// Every tensor's (shape, ggml dtype, byte size) by name, from the header alone. + /// + /// The same sizing and refusals as [`Self::get_tensor_raw`], with each tensor's + /// extent checked against `file_len` instead of the bytes in memory, so a reader + /// from [`Self::header_from_file`] refuses a truncated file exactly as a whole-file + /// one does, without reading any tensor data (#4520 step 2). + pub fn tensor_extents( + &self, + file_len: u64, + ) -> Result, u32, usize)>> { + let file_len = usize::try_from(file_len).unwrap_or(usize::MAX); + let mut result = BTreeMap::new(); + for meta in &self.tensors { + let (_, byte_size, shape) = self.tensor_extent(meta, file_len)?; + result.insert(meta.name.clone(), (shape, meta.dtype, byte_size)); + } + Ok(result) + } + /// (start, byte size, shape) of one tensor, refused when it ends past `limit`. + fn tensor_extent( + &self, + meta: &GgufTensorMeta, + limit: usize, + ) -> Result<(usize, usize, Vec)> { + let name = meta.name.as_str(); let shape: Vec = meta.dims.iter().map(|&d| d as usize).collect(); // BUG-GGUF-002 FIX: Use checked multiplication to prevent integer overflow @@ -260,14 +290,13 @@ impl GgufReader { ), })?; - if tensor_start + byte_size > self.data.len() { + if tensor_start + byte_size > limit { return Err(AprenderError::FormatError { message: format!("Tensor '{name}' data exceeds file size"), }); } - let bytes = self.data[tensor_start..tensor_start + byte_size].to_vec(); - Ok((bytes, shape, meta.dtype)) + Ok((tensor_start, byte_size, shape)) } /// Get all tensors as raw bytes (preserves quantization) diff --git a/crates/aprender-core/src/format/lint/lint.rs b/crates/aprender-core/src/format/lint/lint.rs index 780c7ba032..09b3f661ed 100644 --- a/crates/aprender-core/src/format/lint/lint.rs +++ b/crates/aprender-core/src/format/lint/lint.rs @@ -184,7 +184,19 @@ fn lint_safetensors_file(path: &Path) -> Result { let mut info = ModelLintInfo::default(); - let data = std::fs::read(path)?; + // #3761: the metadata is in the JSON header; read the length and the header, never the + // tensor data (the tensors come from `mapped`). A header past the shared cap is not read. + let head = crate::format::prefix::read_prefix(path, 8)?; + let header_end = head + .get(0..8) + .and_then(|b| b.try_into().ok()) + .map(u64::from_le_bytes) + .and_then(|n| usize::try_from(n).ok()?.checked_add(8)) + .filter(|&n| n <= crate::format::prefix::HEADER_READ_CAP); + let data = match header_end { + Some(n) => crate::format::prefix::read_prefix(path, n)?, + None => head, + }; extract_safetensors_metadata(&data, &mut info); collect_safetensors_tensors(&mapped, &mut info); @@ -318,10 +330,11 @@ fn lint_apr_v1_file(path: &Path) -> Result { /// Lint an APR v2 file (APR\0 or APR2 magic) fn lint_apr_v2_file(path: &Path) -> Result { use crate::format::v2::AprV2Reader; - use std::fs; - // Read file and create reader - let data = fs::read(path)?; + // Read the header + metadata + tensor index and create the reader (#3761: the lint reads + // metadata and index entries only, never the tensor data) + let data = crate::format::prefix::apr_v2_header_prefix(path) + .map_err(|message| crate::error::AprenderError::FormatError { message })?; let reader = AprV2Reader::from_bytes(&data).map_err(|e| crate::error::AprenderError::FormatError { message: format!("Failed to parse APR v2: {e}"), diff --git a/crates/aprender-core/src/format/mod.rs b/crates/aprender-core/src/format/mod.rs index 99f113bdff..32e6ae4325 100644 --- a/crates/aprender-core/src/format/mod.rs +++ b/crates/aprender-core/src/format/mod.rs @@ -83,6 +83,13 @@ pub mod compare; // Re-exported here so `aprender::format::v2::*` keeps resolving unchanged. pub use apr_format::v2; +// #3750 / #3761: the one bounded-prefix policy for header-only reads of a model file +// (apr-format's), and the SafeTensors header reader beside its format. +pub mod prefix; + +#[cfg(test)] +mod prefix_rss_tests; + // APR v2 dequantizing accessor (`get_tensor_as_f32`) re-attached as an // extension trait — the GGUF Q4_K/Q6_K dequant + f16-scaled Q4 path is // framework/quant concern that was SEVERED from the sovereign leaf (#2231). diff --git a/crates/aprender-core/src/format/onnx/reader.rs b/crates/aprender-core/src/format/onnx/reader.rs index 30f2c9885d..f5a052d0af 100644 --- a/crates/aprender-core/src/format/onnx/reader.rs +++ b/crates/aprender-core/src/format/onnx/reader.rs @@ -464,7 +464,8 @@ pub fn is_onnx_file(path: &Path) -> bool { } // Check protobuf magic (ONNX starts with varint tag for field 1, wire type 0) // Field 1 (ir_version) with varint wire type = tag byte 0x08 - std::fs::read(path).is_ok_and(|data| data.len() > 4 && data[0] == 0x08) + // #3761: 5 bytes decide it (`len > 4` and the first byte), never the whole file + crate::format::prefix::read_prefix(path, 5).is_ok_and(|data| data.len() > 4 && data[0] == 0x08) } /// Check if a file is a NeMo archive (.nemo = tar.gz) diff --git a/crates/aprender-core/src/format/prefix.rs b/crates/aprender-core/src/format/prefix.rs new file mode 100644 index 0000000000..72d93947e4 --- /dev/null +++ b/crates/aprender-core/src/format/prefix.rs @@ -0,0 +1,62 @@ +//! Bounded reads of a model file's head (#3750, #3761): the ONE policy (first read + cap) and +//! the APR v2 header reader come from apr-format; the SafeTensors header reader lives here, +//! beside the SafeTensors format. + +pub use apr_format::prefix::*; + +use std::path::Path; + +/// The bytes of a SafeTensors file before its tensor data: the 8-byte little-endian header +/// length and the JSON header it counts. Bounded by [`HEADER_READ_CAP`]. +pub fn safetensors_header_prefix(path: &Path) -> Result, String> { + let head = read_prefix(path, 8).map_err(|e| format!("cannot read {}: {e}", path.display()))?; + let len_bytes: [u8; 8] = head.as_slice().try_into().map_err(|_| { + format!( + "{} is shorter than a SafeTensors length prefix", + path.display() + ) + })?; + let n = usize::try_from(u64::from_le_bytes(len_bytes)) + .ok() + .and_then(|h| h.checked_add(8)) + .filter(|&n| n <= HEADER_READ_CAP) + .ok_or_else(|| { + format!( + "SafeTensors header length {} is past the {} MiB header cap; refused rather than read whole", + u64::from_le_bytes(len_bytes), + HEADER_READ_CAP >> 20 + ) + })?; + read_prefix(path, n).map_err(|e| format!("cannot read {}: {e}", path.display())) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + + fn file_with(bytes: &[u8]) -> tempfile::NamedTempFile { + let mut f = tempfile::NamedTempFile::new().expect("temp file"); + f.write_all(bytes).expect("write"); + f + } + + #[test] + fn a_safetensors_prefix_is_the_length_and_the_json_only() { + let json = br#"{"w":{"dtype":"F32","shape":[1],"data_offsets":[0,4]}}"#; + let mut bytes = (json.len() as u64).to_le_bytes().to_vec(); + bytes.extend_from_slice(json); + bytes.extend_from_slice(&[0u8; 4096]); // tensor data + let f = file_with(&bytes); + let prefix = safetensors_header_prefix(f.path()).expect("the header"); + assert_eq!(prefix.len(), 8 + json.len()); + assert_eq!(&prefix[8..], json); + } + + #[test] + fn a_safetensors_length_past_the_cap_is_refused() { + let f = file_with(&u64::MAX.to_le_bytes()); + let e = safetensors_header_prefix(f.path()).expect_err("an absurd length"); + assert!(e.contains("refused rather than read whole"), "{e}"); + } +} diff --git a/crates/aprender-core/src/format/prefix_rss_tests.rs b/crates/aprender-core/src/format/prefix_rss_tests.rs new file mode 100644 index 0000000000..54a6da354d --- /dev/null +++ b/crates/aprender-core/src/format/prefix_rss_tests.rs @@ -0,0 +1,187 @@ +//! #3761 case rows: aprender-core's header-only readers keep PEAK RSS small on 2 GiB model files. +//! +//! Each fixture is a real file (GGUF, APR v2, APR v1, SafeTensors) extended SPARSELY to 2 GiB. +//! A row runs its readers in a CHILD process (this test binary, re-run on one probe test), +//! because `cargo test` shares one process between tests and its peak would be theirs. The +//! helpers here are shared with the converter's row (`converter::export::rss_tests`). + +use std::io::Write; +use std::path::Path; + +/// Every row's sparse fixture is this long. +pub(crate) const FIXTURE_LEN: u64 = 2 << 30; + +/// Measured (x86-64 debug test binary, 2026-09-21): the format row's peak was 36,916-38,000 KiB over +/// eight runs; the converter row's, 16,200-17,648 KiB. With a whole-file read put back into +/// any one of twelve readers (`list_tensors` on GGUF and on APR v1, lint on SafeTensors and on +/// APR, rosetta `inspect`, `is_onnx_file`, `safetensors_header_prefix`, and the converter's +/// five), its row read 2,110,308-4,218,352 KiB. Put a whole-file read +/// back into any one reader and a row reads over 2 GiB. The bound sits far between. +pub(crate) const PEAK_RSS_BOUND_KB: u64 = 256 * 1024; + +/// `bytes`, then extended SPARSELY to [`FIXTURE_LEN`]: a multi-GiB model that costs no disk. +pub(crate) fn sparse(bytes: &[u8], suffix: &str) -> tempfile::NamedTempFile { + let mut f = tempfile::NamedTempFile::with_suffix(suffix).expect("temp file"); + f.write_all(bytes).expect("write"); + f.as_file().set_len(FIXTURE_LEN).expect("extend sparsely"); + f +} + +pub(crate) fn gguf_bytes() -> Vec { + use crate::format::gguf::{export_tensors_to_gguf, GgmlType, GgufTensor, GgufValue}; + let tensors = vec![GgufTensor { + name: "token_embd.weight".into(), + shape: vec![4, 8], + dtype: GgmlType::F32, + data: vec![0u8; 128], + }]; + let metadata = vec![( + "general.architecture".to_string(), + GgufValue::String("llama".into()), + )]; + let mut bytes = Vec::new(); + export_tensors_to_gguf(&mut bytes, &tensors, &metadata).expect("write GGUF"); + bytes +} + +/// An APR v2 file with an architecture and a user `source_metadata` key in its metadata, so +/// the converter's readers each have something to find. +pub(crate) fn apr_bytes() -> Vec { + use crate::format::v2::{AprV2Metadata, AprV2Writer, TensorDType}; + let mut metadata = AprV2Metadata::new("test"); + metadata.architecture = Some("llama".to_string()); + metadata.custom.insert( + "source_metadata".to_string(), + serde_json::json!({ "origin": "fixture" }), + ); + let mut writer = AprV2Writer::new(metadata); + writer.add_tensor("w", TensorDType::F32, vec![4, 4], vec![0u8; 64]); + let mut bytes = Vec::new(); + writer.write_to(&mut bytes).expect("write APR"); + bytes +} + +/// An APR v1 file: the 32-byte header ("APRN", the metadata length at offset 8) and a JSON +/// metadata section naming one tensor shape. +pub(crate) fn apr_v1_bytes() -> Vec { + let metadata = br#"{"tensor_shapes":{"w":[4,4]}}"#; + let mut bytes = vec![0u8; crate::format::HEADER_SIZE]; + bytes[0..4].copy_from_slice(b"APRN"); + bytes[8..12].copy_from_slice(&(metadata.len() as u32).to_le_bytes()); + bytes.extend_from_slice(metadata); + bytes +} + +pub(crate) fn safetensors_bytes() -> Vec { + let json = br#"{"w":{"dtype":"F32","shape":[1],"data_offsets":[0,4]}}"#; + let mut bytes = (json.len() as u64).to_le_bytes().to_vec(); + bytes.extend_from_slice(json); + bytes.extend_from_slice(&[0u8; 4]); + bytes +} + +/// Run the probe test `probe` in a child with `envs` set, and return the peak RSS (KiB) it +/// reported. Panics, with the child's output, when it reports none. +pub(crate) fn child_peak_kb(probe: &str, envs: &[(&str, &Path)]) -> u64 { + let mut cmd = std::process::Command::new(std::env::current_exe().expect("this test binary")); + cmd.args(["--exact", probe, "--nocapture", "--test-threads=1"]); + for (k, v) in envs { + cmd.env(k, v); + } + let out = cmd.output().expect("run the probe"); + let text = String::from_utf8_lossy(&out.stdout); + let hwm = text + .lines() + .find_map(|l| l.split("PEAK_RSS_KB=").nth(1)) // libtest prints it after the test name + .and_then(|v| v.split_whitespace().next()?.parse::().ok()) + .unwrap_or_else(|| { + panic!( + "the probe {probe} reported no peak: {text}{}", + String::from_utf8_lossy(&out.stderr) + ) + }); + eprintln!("{probe}: peak RSS {hwm} KiB"); + hwm +} + +/// The probe side: print this process's peak RSS where [`child_peak_kb`] reads it. +pub(crate) fn report_peak() { + let status = std::fs::read_to_string("/proc/self/status").expect("/proc/self/status"); + let hwm = status + .lines() + .find_map(|l| l.strip_prefix("VmHWM:")) + .and_then(|v| v.trim().trim_end_matches("kB").trim().parse::().ok()) + .expect("VmHWM in /proc/self/status"); + println!("PEAK_RSS_KB={hwm}"); +} + +const PROBE_GGUF: &str = "APR_3761_CORE_PROBE_GGUF"; +const PROBE_APR: &str = "APR_3761_CORE_PROBE_APR"; +const PROBE_APR_V1: &str = "APR_3761_CORE_PROBE_APR_V1"; +const PROBE_ST: &str = "APR_3761_CORE_PROBE_ST"; + +#[cfg(target_os = "linux")] +#[test] +fn header_only_readers_keep_peak_rss_small_on_2_gib_models() { + let gguf = sparse(&gguf_bytes(), ".gguf"); + let apr = sparse(&apr_bytes(), ".apr"); + let apr_v1 = sparse(&apr_v1_bytes(), ".apr"); + let st = sparse(&safetensors_bytes(), ".safetensors"); + let hwm = child_peak_kb( + "format::prefix_rss_tests::peak_rss_probe", + &[ + (PROBE_GGUF, gguf.path()), + (PROBE_APR, apr.path()), + (PROBE_APR_V1, apr_v1.path()), + (PROBE_ST, st.path()), + ], + ); + assert!( + hwm < PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB on 2 GiB models (bound {PEAK_RSS_BOUND_KB} KiB): a whole-file read is back" + ); +} + +/// Not a test on its own: with the probe variables unset it does nothing. +#[cfg(target_os = "linux")] +#[test] +fn peak_rss_probe() { + let (Some(gguf), Some(apr), Some(apr_v1), Some(st)) = ( + std::env::var_os(PROBE_GGUF), + std::env::var_os(PROBE_APR), + std::env::var_os(PROBE_APR_V1), + std::env::var_os(PROBE_ST), + ) else { + return; + }; + let (gguf, apr, apr_v1, st) = ( + Path::new(&gguf), + Path::new(&apr), + Path::new(&apr_v1), + Path::new(&st), + ); + let options = crate::format::TensorListOptions::default; + let listed = crate::format::list_tensors(gguf, options()).expect("list the GGUF tensors"); + let listed_v1 = + crate::format::list_tensors(apr_v1, options()).expect("list the APR v1 tensors"); + let st_header = crate::format::prefix::safetensors_header_prefix(st).expect("the header"); + crate::format::lint::lint_model_file(apr).expect("lint the APR"); + crate::format::lint::lint_model_file(st).expect("lint the SafeTensors"); + crate::format::rosetta::RosettaStone::new() + .inspect(apr) + .expect("inspect the APR"); + // #4520 step 2: `apr inspect` on a GGUF read the whole file and copied every tensor. + let inspected_gguf = crate::format::rosetta::RosettaStone::new() + .inspect(gguf) + .expect("inspect the GGUF"); + let onnx = crate::format::onnx::is_onnx_file(gguf); + // Reported before the result checks, so a whole-file read fails the row on its peak + report_peak(); + assert_eq!(listed.tensor_count, 1); + assert_eq!(inspected_gguf.tensors.len(), 1); + assert_eq!(inspected_gguf.tensors[0].size_bytes, 128); + assert_eq!(listed_v1.format_version, "v1"); + assert_eq!(listed_v1.tensor_count, 1); + assert_eq!(&st_header[8..9], b"{"); + assert!(!onnx); +} diff --git a/crates/aprender-core/src/format/rosetta/validate_inspect.rs b/crates/aprender-core/src/format/rosetta/validate_inspect.rs index cb971e9d2e..27f1b7f84b 100644 --- a/crates/aprender-core/src/format/rosetta/validate_inspect.rs +++ b/crates/aprender-core/src/format/rosetta/validate_inspect.rs @@ -417,28 +417,31 @@ impl RosettaStone { // ------------------------------------------------------------------------ fn inspect_gguf(&self, path: &Path, file_size: usize) -> Result { - use crate::format::gguf::{load_gguf_raw, GgufRawTensor}; + use crate::format::gguf::{gguf_raw_metadata, GgufReader}; - let result = load_gguf_raw(path)?; + // #4520 step 2 / #3761: header only. `load_gguf_raw` read the whole file and + // copied every tensor (16.7 GB read, 32.8 GB peak RSS on a 27B Q4_K_M) to report + // sizes and metadata the header holds. Same sizing and refusals, no tensor bytes. + let (reader, file_len) = GgufReader::header_from_file(path)?; + let extents = reader.tensor_extents(file_len)?; // Contract: apr-inspect-metadata-propagation-v1 F-INSPECT-META-001 (paiml/aprender#622). // Surface ALL on-disk GGUF KV pairs using their authentic keys (e.g., qwen2.embedding_length, // general.architecture, tokenizer.ggml.model). Previously this was a 4-key hand-written stub // that fabricated ML-shorthand names (n_embd, n_heads, n_layers) — see Five Whys in the // contract YAML for full root-cause analysis. - let meta_map: BTreeMap = result.raw_metadata.clone(); + let meta_map: BTreeMap = gguf_raw_metadata(&reader); // Contract: apr-inspect-dtype-naming-v1 F-INSPECT-DTYPE-001 (paiml/aprender#619). // Render GGML dtype as a human-readable name (F32, Q4_K, Q6_K, …), not the raw u32 // discriminant. Delegates to the same lookup used by `apr tensors` for cross-cmd parity. - let tensors: Vec = result - .tensors - .iter() - .map(|(name, t): (&String, &GgufRawTensor)| TensorInfo { - name: name.clone(), - dtype: crate::format::tensors::ggml_dtype_name(t.dtype).to_string(), - shape: t.shape.clone(), - size_bytes: t.data.len(), + let tensors: Vec = extents + .into_iter() + .map(|(name, (shape, dtype, size_bytes))| TensorInfo { + name, + dtype: crate::format::tensors::ggml_dtype_name(dtype).to_string(), + shape, + size_bytes, stats: None, }) .collect(); @@ -448,7 +451,7 @@ impl RosettaStone { .map(|t| t.shape.iter().product::()) .sum(); - let architecture = result.model_config.architecture.clone(); + let architecture = reader.architecture(); // Contract: apr-inspect-quantization-v1 F-INSPECT-QUANT-001 (paiml/aprender#603). // The model's "quantization" is the dominant dtype by parameter count among its WEIGHT @@ -548,8 +551,9 @@ impl RosettaStone { fn inspect_apr(&self, path: &Path, file_size: usize) -> Result { use crate::format::v2::AprV2Reader; - // Read file into bytes - let data = std::fs::read(path).map_err(|e| AprenderError::FormatError { + // Read the header + metadata + tensor index (#3761: inspect reports metadata and index + // entries with no stats, so the tensor data is never read) + let data = crate::format::prefix::apr_v2_header_prefix(path).map_err(|e| AprenderError::FormatError { message: format!("Cannot read APR file: {e}"), })?; diff --git a/crates/aprender-core/src/format/safetensors.rs b/crates/aprender-core/src/format/safetensors.rs index c042174e70..0baf894d93 100644 --- a/crates/aprender-core/src/format/safetensors.rs +++ b/crates/aprender-core/src/format/safetensors.rs @@ -18,14 +18,32 @@ fn ggml_dtype_element_size(dtype: u32) -> f64 { /// List tensors from GGUF file bytes fn list_tensors_gguf(data: &[u8], options: TensorListOptions) -> Result { - // #3661: a FormatError contributes its message, not its Display, or the - // result reads "Invalid model format: Failed to parse GGUF: Invalid model format: …". - let reader = GgufReader::from_bytes(data.to_vec()).map_err(|e| AprenderError::FormatError { - message: match e { - AprenderError::FormatError { message } => format!("Failed to parse GGUF: {message}"), - other => format!("Failed to parse GGUF: {other}"), - }, - })?; + let reader = GgufReader::from_bytes(data.to_vec()).map_err(|e| gguf_parse_error(&error_message(e)))?; + list_tensors_gguf_reader(&reader, data.len() as u64, options) +} + +/// #3661: a FormatError contributes its message, not its Display, or the +/// result reads "Invalid model format: Failed to parse GGUF: Invalid model format: …". +fn error_message(e: AprenderError) -> String { + match e { + AprenderError::FormatError { message } => message, + other => other.to_string(), + } +} + +fn gguf_parse_error(message: &str) -> AprenderError { + AprenderError::FormatError { + message: format!("Failed to parse GGUF: {message}"), + } +} + +/// List tensors from a parsed GGUF. `file_len` is the whole file's length, which the +/// #2569 extent check holds every tensor to, whether `reader` holds the file or its header. +fn list_tensors_gguf_reader( + reader: &GgufReader, + file_len: u64, + options: TensorListOptions, +) -> Result { // #2569: every row below asserts that `size_bytes` of tensor data exist at a // declared offset. Prove that before printing it. Run over ALL tensors, ahead @@ -39,7 +57,7 @@ fn list_tensors_gguf(data: &[u8], options: TensorListOptions) -> Result Option> { + use std::io::Read; + let mut buf = Vec::with_capacity(4); + fs::File::open(path) + .ok()? + .take(4) + .read_to_end(&mut buf) + .ok()?; + Some(buf) +} + /// Check if a file is a valid .apr v2 file pub fn is_apr_file>(path: P) -> bool { - fs::read(path.as_ref()).is_ok_and(|data| data.len() >= 4 && data[0..4] == MAGIC) + read_magic(path.as_ref()).is_some_and(|data| data.len() >= 4 && data[0..4] == MAGIC) } /// Detect model format from file extension @@ -322,7 +334,7 @@ fn format_from_extension(path: &Path) -> Option<&'static str> { /// Detect model format from file magic bytes fn format_from_magic(path: &Path) -> &'static str { - let Ok(data) = fs::read(path) else { + let Some(data) = read_magic(path) else { return "unknown"; }; if data.len() < 4 { @@ -347,3 +359,7 @@ pub fn detect_format>(path: P) -> &'static str { } include!("helpers_tests.rs"); + +#[cfg(all(test, target_os = "linux"))] +#[path = "helpers_rss_tests.rs"] +mod rss_tests; diff --git a/crates/aprender-serve/src/apr/helpers_rss_tests.rs b/crates/aprender-serve/src/apr/helpers_rss_tests.rs new file mode 100644 index 0000000000..554ebf8512 --- /dev/null +++ b/crates/aprender-serve/src/apr/helpers_rss_tests.rs @@ -0,0 +1,70 @@ +//! #3761 case row: `is_apr_file` and `detect_format` (by magic) read 4 bytes of a 2 GiB file, +//! never the whole of it. Each ran `fs::read` on the model to look at its first four bytes. +//! +//! The probe runs in a CHILD process (this test binary, re-run on `peak_rss_probe`), because +//! `cargo test` shares one process between tests and its peak would be theirs. +//! +//! Measured (x86-64 debug, 2026-09-21): 19.9-20.3 MiB over three runs; with `fs::read` put back +//! into `is_apr_file` or `format_from_magic`, 2,113,676 / 2,112,484 KiB. + +use std::io::Write; +use std::path::Path; + +const PROBE: &str = "APR_3761_SERVE_MAGIC_PROBE"; + +/// Far above a 4-byte read, far below a 2 GiB one. +const PEAK_RSS_BOUND_KB: u64 = 256 * 1024; + +#[test] +fn magic_checks_of_a_2_gib_file_keep_peak_rss_small() { + // An APR magic, extended sparsely to 2 GiB, with no extension so `detect_format` reads it + let mut f = tempfile::NamedTempFile::new().expect("temp file"); + f.write_all(b"APR\0").expect("write"); + f.as_file().set_len(2 << 30).expect("extend sparsely"); + let out = std::process::Command::new(std::env::current_exe().expect("this test binary")) + .args([ + "--exact", + "apr::helpers::rss_tests::peak_rss_probe", + "--nocapture", + "--test-threads=1", + ]) + .env(PROBE, f.path()) + .output() + .expect("run the probe"); + let text = String::from_utf8_lossy(&out.stdout); + let hwm = text + .lines() + .find_map(|l| l.split("PEAK_RSS_KB=").nth(1)) + .and_then(|v| v.split_whitespace().next()?.parse::().ok()) + .unwrap_or_else(|| { + panic!( + "the probe reported no peak: {text}{}", + String::from_utf8_lossy(&out.stderr) + ) + }); + eprintln!("apr::helpers::rss_tests::peak_rss_probe: peak RSS {hwm} KiB"); + assert!( + hwm < PEAK_RSS_BOUND_KB, + "peak RSS {hwm} KiB checking the magic of a 2 GiB file (bound {PEAK_RSS_BOUND_KB} KiB): a whole-file read is back" + ); +} + +/// Not a test on its own: with the probe variable unset it does nothing. +#[test] +fn peak_rss_probe() { + let Some(path) = std::env::var_os(PROBE) else { + return; + }; + let path = Path::new(&path); + let is_apr = super::is_apr_file(path); + let format = super::detect_format(path); + let status = std::fs::read_to_string("/proc/self/status").expect("/proc/self/status"); + let hwm = status + .lines() + .find_map(|l| l.strip_prefix("VmHWM:")) + .and_then(|v| v.trim().trim_end_matches("kB").trim().parse::().ok()) + .expect("VmHWM in /proc/self/status"); + println!("PEAK_RSS_KB={hwm}"); + assert!(is_apr, "the fixture carries the APR magic"); + assert_eq!(format, "apr"); +} diff --git a/docs/audits/3750-whole-file-model-reads.md b/docs/audits/3750-whole-file-model-reads.md index 7347459779..e4b3e1769a 100644 --- a/docs/audits/3750-whole-file-model-reads.md +++ b/docs/audits/3750-whole-file-model-reads.md @@ -5,13 +5,15 @@ Measured on `PMAT-3750-qa-header-only-reads` (base `52f43da71`) with **231 hits** after this PR: **99 production**, **132 test-only**, plus the **13 `apr qa` reads this PR converted** (they no longer match). Verdicts: -- **CONVERTED** — read only the header or magic now (this PR). +- **CONVERTED** — reads only the header or magic now: the `apr qa` path in #3750 PR A, the 17 sites marked `CONVERTED (#3761)` in PR B. - **PR-B …** — needs only the magic / header / metadata and is converted by the 0.69.1 sub-issue #3761 (#3750 PR B), which puts the APR and SafeTensors prefix readers in their format crates. - **PR-B STREAMED** — needs every byte, but never all of them at once; #3761 streams it. - **WHOLE-DATA** — consumes the tensor data or every byte (loaders, converters, validators, copies, uploads); "could stream" notes a whole-file buffer that is not needed all at once. - **NOT-MODEL** — reads a file that is not a model. **DOC** — a doc example, not executed code. -Counts over the 99 production sites: DOC 9, NOT-MODEL 11, PR-B CONDITIONAL 1, PR-B HEADER-ONLY 11, PR-B MAGIC-ONLY 4, PR-B STREAMED 1, WHOLE-DATA 62. +Counts over the 99 production sites: DOC 9, NOT-MODEL 11, CONVERTED (#3761) 17 (they were PR-B CONDITIONAL 1, HEADER-ONLY 11, MAGIC-ONLY 4, STREAMED 1), WHOLE-DATA 62. + +After #3761 the same `git grep` finds **215 hits**: the 17 are gone, and one line matches that is not a whole-file read (aprender-serve's `read_magic`, a `take(4)` before `read_to_end`). The rows below keep their line numbers on `52f43da71`. ## Converted by this PR (the `apr qa` path, line numbers on `52f43da71`) @@ -67,13 +69,38 @@ What the numbers say: tracks streaming it from a map. - 36,278,440 − 19,367,800 = 16,910,640 KiB (16.1 GiB) less peak per `apr qa` run on this model. +## Converted by #3761 (#3750 PR B) + +ONE bounded-prefix policy: `apr_format::prefix` (first read 16 MiB, doubling, 256 MiB cap, refused +past it, never read whole) with the APR v2 header reader; aprender-core's `format::prefix` re-exports it +and adds the SafeTensors header reader beside its format; apr-cli's `model_header.rs` delegates to it. + +Each reader has a case row: a real file extended SPARSELY to 2 GiB, the reader run in a CHILD process +(the test binary re-run on one probe test), its VmHWM held under a measured bound. With a whole-file read +put back into any one reader, its row goes RED: + +| row | readers | peak (KiB) | with a whole-file read back (KiB) | +|---|---|---|---| +| `apr_format::prefix::tests` | `apr_v2_header_prefix` | 4,608 | 2,100,864 | +| `aprender::format::prefix_rss_tests` | `list_tensors` (GGUF, APR v1), `safetensors_header_prefix`, lint (SafeTensors, APR), rosetta `inspect`, `is_onnx_file` | 36,916-38,000 | 2,116,876-4,218,352 | +| `aprender::format::converter::export::rss_tests` | the converter's five APR readers | 16,200-17,648 | 2,110,308-4,208,572 | +| `realizar::apr::helpers::rss_tests` | `is_apr_file`, `detect_format` by magic | 20,352-20,736 | 2,112,484-2,113,676 | +| `apr-cli model_header::tests` | the qa header readers + `gguf_header_bytes` | 43,368-44,856 | 2,135,696 | +| `apr-cli embed_viz::tests` | `gguf_vocab` | 40,768-43,304 | 2,116,644-2,117,924 | +| `apr-cli eval_mod_tests` | `count_safetensors_keys`, `verify_single_file` | 25,900-27,952 | 2,114,024-2,115,124 | +| `apr-cli bench::rss_tests` | `apr bench` on GGUF | 2,123,680-2,125,672 (one map: the bench runs the model) | 4,212,152-4,212,352 | + +The rows found two copies the `git grep` could not: `list_tensors_gguf` copied the bytes it was given +(`data.to_vec()`), and `apr bench` mapped the model in `run_gguf_benchmark` and then again in each backend +path. realizar's map pre-faults every page (MAP_POPULATE, PMAT-304), so two maps count the file twice. + ## Every production site | site | function | verdict | what the bytes are used for | |---|---|---|---| | apr-cli/src/commands/audio_inspect.rs:108 | inspect | NOT-MODEL | a WAV file; parses its fmt/data chunks | -| apr-cli/src/commands/benchmark.rs:104 | run_realizar_benchmark | PR-B MAGIC-ONLY | bytes used only for detect_format on the first 8 | -| apr-cli/src/commands/benchmark.rs:167 | run_gguf_benchmark | PR-B HEADER-ONLY | GGUFModel::from_bytes only for the tokenizer; the model is mapped separately | +| apr-cli/src/commands/benchmark.rs:104 | run_realizar_benchmark | CONVERTED (#3761) | bytes used only for detect_format on the first 8. **Now:** 8 bytes (`read_prefix`). Row: `bench::rss_tests` | +| apr-cli/src/commands/benchmark.rs:167 | run_gguf_benchmark | CONVERTED (#3761) | GGUFModel::from_bytes only for the tokenizer; the model is mapped separately. **Now:** the ONE map the bench runs on (`MappedGGUFModel`), handed to the CPU, CUDA and MoE paths, which each mapped the file again. Row: `bench::rss_tests`, bound = one map + 256 MiB | | apr-cli/src/commands/canary.rs:166 | load_tensor_data_gguf | WHOLE-DATA | loads every GGUF tensor as f32 for the canary | | apr-cli/src/commands/canary.rs:186 | load_tensor_data_apr | WHOLE-DATA | loads every APR tensor as f32 for the canary | | apr-cli/src/commands/chat_session_02.rs:21 | new | WHOLE-DATA | the chat session keeps the model bytes and runs inference from them | @@ -82,10 +109,10 @@ What the numbers say: | apr-cli/src/commands/distill.rs:695 | run_cuda_backend | WHOLE-DATA | teacher model weights for distillation | | apr-cli/src/commands/distill.rs:738 | run_cuda_backend | WHOLE-DATA | student model weights for distillation | | apr-cli/src/commands/embed.rs:245 | run | WHOLE-DATA | APR v2 reader; reads the embedding weights | -| apr-cli/src/commands/embed_viz.rs:382 | gguf_vocab | PR-B HEADER-ONLY | LlamaTokenizer::from_gguf_bytes needs only the header vocabulary | +| apr-cli/src/commands/embed_viz.rs:382 | gguf_vocab | CONVERTED (#3761) | LlamaTokenizer::from_gguf_bytes needs only the header vocabulary. **Now:** the whole header, cut at `data_offset` (`gguf_header_bytes`). Row: `embed_viz::tests::gguf_vocab_of_a_2_gib_gguf_keeps_peak_rss_small` | | apr-cli/src/commands/embed_viz_lint.rs:35 | run | NOT-MODEL | an output artifact whose determinism is classified | -| apr-cli/src/commands/eval/mod.rs:746 | count_safetensors_keys | PR-B HEADER-ONLY | SafeTensors: 8-byte length + JSON header, counts keys | -| apr-cli/src/commands/eval/mod.rs:922 | verify_single_file | PR-B STREAMED | SafeTensors header checks (bounded), plus an FNV-1a hash of EVERY byte: it needs the whole file, never all of it at once, so #3761 streams the hash in 1 MiB chunks | +| apr-cli/src/commands/eval/mod.rs:746 | count_safetensors_keys | CONVERTED (#3761) | SafeTensors: 8-byte length + JSON header, counts keys. **Now:** `safetensors_header_prefix`. Row: `eval_mod_tests::safetensors_readers_of_a_2_gib_file_keep_peak_rss_small` | +| apr-cli/src/commands/eval/mod.rs:922 | verify_single_file | CONVERTED (#3761) | SafeTensors header checks (bounded), plus an FNV-1a hash of EVERY byte: it needs the whole file, never all of it at once, so #3761 streams the hash in 1 MiB chunks. **Now:** 8 bytes, then the JSON header when its size is valid, and the hash STREAMED in 1 MiB chunks (`compute_file_hash_streamed`, equal to the whole-buffer hash by test). Row: as above | | apr-cli/src/commands/eval/mod.rs:1314 | run_encrypt | WHOLE-DATA | encrypts every byte (could stream) | | apr-cli/src/commands/eval/mod.rs:1399 | run_decrypt | WHOLE-DATA | decrypts every byte (could stream) | | apr-cli/src/commands/eval/mod.rs:1478 | derive_encryption_key | NOT-MODEL | an encryption key file | @@ -118,25 +145,25 @@ What the numbers say: | aprender-core/src/bundle/mmap.rs:204 | open | WHOLE-DATA | loads the whole bundle (named mmap, reads; could map) | | aprender-core/src/cluster/kmeans_impl.rs:110 | load | WHOLE-DATA | deserializes a small classical model | | aprender-core/src/ensemble/moe.rs:323 | load | WHOLE-DATA | deserializes the ensemble | -| aprender-core/src/format/converter/apr_export_fn.rs:133 | detect_apr_architecture_for_completeness | PR-B HEADER-ONLY | APR metadata architecture only | +| aprender-core/src/format/converter/apr_export_fn.rs:133 | detect_apr_architecture_for_completeness | CONVERTED (#3761) | APR metadata architecture only. **Now:** `apr_v2_header_prefix`. Row: `converter::export::rss_tests` | | aprender-core/src/format/converter/convert_report.rs:173 | load_apr_tensors_f32 | WHOLE-DATA | loads every APR tensor as f32 | | aprender-core/src/format/converter/gguf_export_config.rs:516 | export_to_gguf | WHOLE-DATA | exports every tensor to GGUF | | aprender-core/src/format/converter/metadata.rs:563 | export_apr_to_gguf_raw | WHOLE-DATA | raw APR -> GGUF export of every tensor | -| aprender-core/src/format/converter/tensor.rs:202 | extract_apr_tokenizer_hint | PR-B HEADER-ONLY | tokenizer hint from the APR metadata section | -| aprender-core/src/format/converter/tensor.rs:226 | read_apr_metadata | PR-B HEADER-ONLY | APR metadata only | -| aprender-core/src/format/converter/tensor.rs:382 | extract_user_metadata | PR-B HEADER-ONLY | user metadata from the APR metadata section | -| aprender-core/src/format/converter/tensor.rs:439 | detect_apr_quantization | PR-B HEADER-ONLY | counts tensor dtypes from the APR tensor index | +| aprender-core/src/format/converter/tensor.rs:202 | extract_apr_tokenizer_hint | CONVERTED (#3761) | tokenizer hint from the APR metadata section. **Now:** APR v2: `apr_v2_header_prefix` (its writer zero-pads the metadata, so the scan's terminator is absent and the answer is None, as it was reading the whole model); any other layout: a growing prefix under the cap. Row: `converter::export::rss_tests`; the legacy path's positive control: `test_infer_tokenizer_json_legacy_apr_with_tokenizer` | +| aprender-core/src/format/converter/tensor.rs:226 | read_apr_metadata | CONVERTED (#3761) | APR metadata only. **Now:** `apr_v2_header_prefix`. Row: `converter::export::rss_tests` | +| aprender-core/src/format/converter/tensor.rs:382 | extract_user_metadata | CONVERTED (#3761) | user metadata from the APR metadata section. **Now:** 24 header bytes, then only up to the end of the metadata section (under the cap). Row: `converter::export::rss_tests` | +| aprender-core/src/format/converter/tensor.rs:439 | detect_apr_quantization | CONVERTED (#3761) | counts tensor dtypes from the APR tensor index. **Now:** `apr_v2_header_prefix`. Row: `converter::export::rss_tests` | | aprender-core/src/format/converter/tokenizer_loader.rs:482 | load_tokenizer_from_sentencepiece | NOT-MODEL | a SentencePiece tokenizer.model | | aprender-core/src/format/core_io.rs:173 | read_file_content | WHOLE-DATA | generic whole-content reader (callers decide) | | aprender-core/src/format/gguf/reader_parsing.rs:6 | from_file | WHOLE-DATA, could stream | GgufReader::from_file owns the whole file by API (importers read tensors). `apr qa`'s tensor_contract gate reaches it through `RosettaStone::validate`: the 19.36 GB heap peak measured above. It reads every tensor, but never needs them all at once (#3790) | | aprender-core/src/format/gguf/reader_parsing.rs:20 | from_file_full | WHOLE-DATA | GgufReader::from_file_full, the shard merge reads tensors | -| aprender-core/src/format/lint/lint.rs:187 | lint_safetensors_file | PR-B HEADER-ONLY | SafeTensors metadata from the header; tensors come from the existing map | -| aprender-core/src/format/lint/lint.rs:324 | lint_apr_v2_file | PR-B HEADER-ONLY | lints APR metadata fields | +| aprender-core/src/format/lint/lint.rs:187 | lint_safetensors_file | CONVERTED (#3761) | SafeTensors metadata from the header; tensors come from the existing map. **Now:** the 8-byte length and the JSON header (under the cap). Row: `format::prefix_rss_tests` | +| aprender-core/src/format/lint/lint.rs:324 | lint_apr_v2_file | CONVERTED (#3761) | lints APR metadata fields. **Now:** `apr_v2_header_prefix`. Row: `format::prefix_rss_tests` | | aprender-core/src/format/onnx/reader.rs:5 | from_file | WHOLE-DATA | parses the ONNX protobuf including initializers | -| aprender-core/src/format/onnx/reader.rs:467 | is_onnx_file | PR-B MAGIC-ONLY | checks data[0] == 0x08 only | +| aprender-core/src/format/onnx/reader.rs:467 | is_onnx_file | CONVERTED (#3761) | checks data[0] == 0x08 only. **Now:** 5 bytes. Row: `format::prefix_rss_tests` | | aprender-core/src/format/rosetta/validate_inspect.rs:7 | validate_apr | WHOLE-DATA | rosetta validate reads tensor data | -| aprender-core/src/format/rosetta/validate_inspect.rs:552 | inspect_apr | PR-B HEADER-ONLY | rosetta inspect: metadata + tensor index entries | -| aprender-core/src/format/safetensors.rs:448 | list_tensors | PR-B CONDITIONAL | GGUF/APR v1 listing: data only with --stats | +| aprender-core/src/format/rosetta/validate_inspect.rs:552 | inspect_apr | CONVERTED (#3761) | rosetta inspect: metadata + tensor index entries. **Now:** `apr_v2_header_prefix`. Row: `format::prefix_rss_tests` | +| aprender-core/src/format/safetensors.rs:448 | list_tensors | CONVERTED (#3761) | GGUF/APR v1 listing: data only with --stats. **Now:** GGUF without `--stats`: the header (growing prefix). It was not only this read: `list_tensors_gguf` then COPIED the bytes (`data.to_vec()`), 4.2 GB peak on the 2 GiB fixture. GGUF with `--stats` and APR v1: a map, which pages in only what is read. Row: `format::prefix_rss_tests` (GGUF and APR v1) | | aprender-core/src/index/persistent_hnsw.rs:110 | open | NOT-MODEL | an HNSW index file | | aprender-core/src/inspect/safetensors.rs:174 | from_file | WHOLE-DATA | keeps the bytes for later tensor reads | | aprender-core/src/linear_model/elastic_net.rs:165 | load | WHOLE-DATA | deserializes a small classical model | @@ -151,8 +178,8 @@ What the numbers say: | aprender-core/src/setfit/import.rs:699 | read_required | NOT-MODEL | a required SetFit sidecar file | | aprender-core/src/tree/classifier.rs:156 | load | WHOLE-DATA | deserializes a small classical model | | aprender-core/src/verify/ground_truth.rs:88 | from_bin_file | NOT-MODEL | a ground-truth .bin | -| aprender-serve/src/apr/helpers.rs:309 | is_apr_file | PR-B MAGIC-ONLY | data[0..4] == MAGIC only | -| aprender-serve/src/apr/helpers.rs:325 | format_from_magic | PR-B MAGIC-ONLY | format from the first 4 bytes only | +| aprender-serve/src/apr/helpers.rs:309 | is_apr_file | CONVERTED (#3761) | data[0..4] == MAGIC only. **Now:** 4 bytes (`read_magic`). Row: `apr::helpers::rss_tests` | +| aprender-serve/src/apr/helpers.rs:325 | format_from_magic | CONVERTED (#3761) | format from the first 4 bytes only. **Now:** 4 bytes (`read_magic`). Row: `apr::helpers::rss_tests` | | aprender-serve/src/apr/loading_mmap.rs:48 | load | WHOLE-DATA | a COMPRESSED .apr must be decompressed whole | | aprender-serve/src/apr/loading_mmap.rs:69 | load | WHOLE-DATA | the wasm32 fallback (no mmap) | | aprender-serve/src/apr_transformer/from_apr_file.rs:32 | from_apr_file | WHOLE-DATA | loads the transformer's weights | diff --git a/docs/audits/quorum-PMAT-3761.json b/docs/audits/quorum-PMAT-3761.json new file mode 100644 index 0000000000..c9d4cb8568 --- /dev/null +++ b/docs/audits/quorum-PMAT-3761.json @@ -0,0 +1,302 @@ +{ + "ticket": "PMAT-3761", + "base": "origin/main", + "base_resolved": "origin/main", + "base_note": "no origin/origin/main exists; judged against the local ref", + "head": "42dba0be643a69ae812fa40090ee00347564376f", + "diff_sha256": "87c1e947a7e8f80bf1420de98969b33ad9719179277fd9cd3d61b323c74ad4a8", + "width": 3, + "executor": "agy", + "prompt_mode": "file", + "prompt_bytes": 122357, + "prompt_sha256": "be28c2b8c111bf09ba228aeb2d8226a01cac55542be22081c2d366ddadb4a262", + "author": { + "model": "claude-opus-5-5", + "family": "claude", + "source": "flag" + }, + "agreed": true, + "lanes": [ + { + "lane": 1, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "test", + "findings": [], + "raw_bytes": 926, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "be28c2b8c111bf09ba228aeb2d8226a01cac55542be22081c2d366ddadb4a262", + "trace": { + "input_sha256": "fb9e696a666f8d832f6fcdd2427cf5b983c119d01a538b9dcda46a33052d84c2", + "output_sha256": "30b287fe3df0ea2a8b839efdf4b54271aca8461d83da8f8be40fecba898771b6", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + }, + { + "lane": 2, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "No refuting evidence found; diff matches ticket PMAT-3761's done_when criteria with tested, memory-safe conversions.", + "findings": [ + { + "file": "N/A", + "claim": "No findings — reviewed diff supports the ticket with tests; issuing PASS.", + "grounding": "cited" + } + ], + "raw_bytes": 1391, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "be28c2b8c111bf09ba228aeb2d8226a01cac55542be22081c2d366ddadb4a262", + "trace": { + "input_sha256": "56eb6e938760867363bc25f441ff3c07969ed5cc4f53ceb603b4d0bf6b7befd2", + "output_sha256": "3f03e0aca59ceafda1b071a56a19c138e093fcbd37f2d3fb03ada9b88cf69f6a", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [ + { + "model": "gemini-3.1-pro-high", + "why": "family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ] + }, + "repeat_of": 1 + }, + { + "lane": 3, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "PMAT-3761 implements all 4 done_when criteria: 17 header-only whole-file reads converted to bounded-prefix policy (16 MiB first, doubling, 256 MiB cap); list_tensors GGUF conditional on --stats; 8 case rows measure peak RSS under 256 MiB bound with 22 mutants specified to turn RED; ledger updated with CONVERTED verdicts. Diff centralizes policy in apr-format/prefix.rs, re-exports via aprender-core, and refactors 17 production sites. Critical pattern: functions now take references to pre-mapped models instead of re-reading. Bonus: apr inspect GGUF header-only (32.8 GB → 75 MB RSS). All acceptance criteria satisfied; no weakened gates or broken tests detected.", + "findings": [], + "raw_bytes": 2242, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-haiku-4-5", + "model_measured": "claude-haiku-4-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "be28c2b8c111bf09ba228aeb2d8226a01cac55542be22081c2d366ddadb4a262", + "trace": { + "input_sha256": "a97cd0832786f8c2e91adfd6bdb85109afebcca810d323cba82386b7ce31924f", + "output_sha256": "28b3d98d6886382dc6468af9562ab42078eb261d30a5889da9edd7cb9245c19d", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-haiku-4-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-haiku-4-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + } + ], + "dissent": [], + "dedup": [ + { + "file": "N/A", + "line": null, + "lanes_agreeing": [ + 2 + ], + "claims": [ + "No findings — reviewed diff supports the ticket with tests; issuing PASS." + ] + } + ], + "uncovered": [], + "coverage_source": "lanes", + "partial": false, + "partial_reasons": [ + "lane models: only 2 distinct model ids across 3 lanes (claude-haiku-4-5, claude-sonnet-5) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)", + "lane 2: judged by claude-sonnet-5 after falling through gemini-3.1-pro-high (skipped: family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24) (PMAT-321)" + ], + "fallback": { + "same_family_width": 2, + "chain": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "gemini-3.1-pro-high", + "family": "gemini", + "disposition": "configured" + }, + { + "model": "claude-haiku-4-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "claude-opus-5-5", + "family": "claude", + "disposition": "excluded-self-review", + "why": "the author's own model id (claude-opus-5-5) — a model never reviews its own diff, at any width (R-15a identity bar)" + }, + { + "model": "qwen3.5", + "family": "qwen", + "disposition": "not-run", + "why": "no quorum.local_lane in the config — the aprender lane has no model to load" + } + ], + "precheck": [ + { + "family": "gemini", + "model": "gemini-3.1-pro-high", + "probe": 0, + "outcome": "policy", + "reason": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "prah": { + "source": "install-receipt", + "path": "/home/noah/.claude/skills/paiml-implement/bin/prah" + }, + "tier1": { + "agy_tier1_only": true, + "tier1": false, + "matched": [], + "paths": 33, + "patterns": [ + "^\\.github/workflows/", + "(^|/)release[^/]*\\.(ya?ml|sh|rs|toml)$", + "cuda|kernel|\\.cu$|\\.ptx$", + "(^|/)[^/]*(gate|guard)[^/]*\\.(sh|rs|py)$|^hooks/", + "security|secret|credential|(^|/)deny\\.toml$", + "^skills/quorum-review/|(^|/)(receipt-lint|roadmap-lint|release-lint|kind-gate|model-gate)[^/]*$|^crates/prah-lint/" + ], + "builtin": "^skills/quorum-review/|(^|/)skills/paiml-implement/config\\.json$|(^|/)(modellib|lane-reduce|lane-fallback|lane-group|agy-lane|cc-lane|receipt-lint|route|quota)\\.sh$|^crates/prah-lint/" + }, + "bucket": { + "ledger": "/home/noah/.local/state/paiml-implement/agy-bucket.jsonl", + "window_s": 18000, + "pace": "off", + "buckets": {}, + "closed": [] + } + }, + "auto_merge": { + "checked": false, + "was_armed": false, + "disarmed": false, + "note": "no --pr given: nothing to disarm" + }, + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "cheap_seat": "claude-haiku-4-5", + "advisory_lane": { + "state": "not-run", + "counts": false, + "row": { + "NotRun": "ContextOverflow" + }, + "verdict": "unavailable", + "why": "the brief is 122357 bytes, over quorum.advisory_lane.max_brief_bytes=24576; a cut brief is a different question, so the lane was not launched", + "served_by": null, + "gpu_proof": null, + "apr": null, + "model": null, + "model_sha256": null, + "rc": 0, + "wall_s": 0, + "collected_s": 0, + "budget_s": 128, + "budget_basis": "ledger: 2 x p95 64 s over 10 gx10-cuda Verdict rows", + "brief": { + "bytes": 122357, + "sent_bytes": 0, + "max_bytes": 24576 + }, + "trace": null, + "attempts": [], + "ledger": "/home/noah/.local/state/paiml-implement/advisory-ledger.jsonl", + "raw": "advisory.json", + "agrees_with_counted": null, + "counted": "PASS" + }, + "lint": { + "ok": true, + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5-5/claude same_family=3/2 degraded=same-family (policy: agy reserved for tier-1 diffs)" + } +} diff --git a/docs/roadmaps/entries/PMAT-3761.yaml b/docs/roadmaps/entries/PMAT-3761.yaml new file mode 100644 index 0000000000..d38cbd1707 --- /dev/null +++ b/docs/roadmaps/entries/PMAT-3761.yaml @@ -0,0 +1,22 @@ +- id: PMAT-3761 + github_issue: 3761 + item_type: task + title: '#3750 PR B: the 17 header-only whole-file model reads outside apr qa — magic/header/metadata readers in their format crates' + status: in_progress + priority: critical + assigned_to: aprender-0e + created: 2026-09-21T21:52:15Z + updated: 2026-09-21T21:52:15Z + spec: null + acceptance_criteria: + - 'THIS PR IS #3750 PR B (#3761, 0.69.1). Cop ruling (aprender-04 to aprender-0e, 2026-09-21), verbatim: "PR B = a sub-issue in **0.69.1**, not later." Stacked on #3750 PR A (PMAT-3750), whose ledger this PR updates.' + - 'Issue #3761 done_when 1, verbatim: "each site above reads only what it uses (a bounded magic read, or the format''s header prefix); the APR v2 prefix reader lives in apr-format and the SafeTensors one in aprender-core, sharing ONE bounded-prefix policy (first read + cap) with apr-cli''s model_header.rs"' + - 'Issue #3761 done_when 2, verbatim: "the `list_tensors` GGUF/APR v1 path reads tensor data only when `--stats` asks for it"' + - 'Issue #3761 done_when 3, verbatim: "a case row per reader proving it never touches tensor data (a sparse multi-GB fixture + a peak-RSS bound, as PR A''s), and a mutant that puts a whole-file read back goes RED"' + - 'Issue #3761 done_when 4, verbatim: "the ledger''s rows are updated to CONVERTED"' + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'Closes #3761; Refs #3750. WHAT LANDS. ONE bounded-prefix policy: crates/apr-format/src/prefix.rs (HEADER_FIRST_READ 16 MiB, doubling, HEADER_READ_CAP 256 MiB, refused past it, never read whole; read_prefix, parse_growing_prefix, apr_v2_header_prefix bounded by the header''s data_offset). crates/aprender-core/src/format/prefix.rs re-exports it and adds safetensors_header_prefix (8-byte length + JSON header, capped) beside the SafeTensors format, as done_when 1 places it. apr-cli model_header.rs delegates to it and adds gguf_header_bytes (the header cut at data_offset, for LlamaTokenizer::from_gguf_bytes, which is not prefix-safe). The 17 sites (ledger rows now CONVERTED (#3761), each naming what it reads now and its case row): bench format detect 8 bytes; bench GGUF maps ONCE and hands the map to the CPU, CUDA and MoE paths (each mapped again before; realizar''s map pre-faults every page, MAP_POPULATE); gguf_vocab via gguf_header_bytes; eval count_safetensors_keys via safetensors_header_prefix; verify_single_file reads the header and STREAMS its FNV-1a hash in 1 MiB chunks (a test proves the streamed hash equals the whole-buffer hash); the converter''s five APR readers via apr_v2_header_prefix or a read bounded by the metadata section; lint SafeTensors (length + header) and APR (apr_v2_header_prefix); rosetta inspect_apr; is_onnx_file 5 bytes; aprender-serve is_apr_file / format_from_magic 4 bytes (read_magic). list_tensors (done_when 2): GGUF without --stats parses the header from a growing prefix; the old path also COPIED the bytes (list_tensors_gguf data.to_vec()), 4.2 GB peak on the 2 GiB fixture. GGUF with --stats and APR v1 use a map, which pages in only what is read. extract_apr_tokenizer_hint: APR v2 input is scanned within apr_v2_header_prefix. The v2 writer zero-pads its metadata, so the terminator this legacy scan looks for is absent and the answer is None, as it was when it read the whole model. Any other layout gets a growing prefix under the cap, and test_infer_tokenizer_json_legacy_apr_with_tokenizer is its positive control. CASE ROWS (done_when 3): each is a real file extended SPARSELY to 2 GiB, the readers run in a CHILD process (the test binary re-run on one probe test), and VmHWM is held under 256 MiB. apr-format 4,608 KiB; aprender-core format row 36.1-37.1 MiB and converter row 15.8-17.2 MiB; aprender-serve 19.9-20.3 MiB; apr-cli model_header 42.4-43.8 MiB, gguf_vocab 39.8-42.3 MiB, eval 25.3-27.3 MiB. apr bench''s bound is one map + 256 MiB, because the bench runs the model: 2,123,680-2,125,672 KiB. MUTANTS (done_when 3): a whole-file read put back into each reader, 22 in all (apr-format 1; aprender-core 12, covering list_tensors GGUF and APR v1, lint x2, inspect, is_onnx_file, safetensors_header_prefix and the converter''s five; aprender-serve 2; apr-cli 7, covering gguf_header_bytes, gguf_vocab, count_safetensors_keys, verify_single_file header and hash, bench format detect and bench tokenizer read). Every one goes RED at 2,100,864-4,218,352 KiB. Each probe reports its peak before its result checks, so a mutant fails on RSS, not on a side assertion. Ledger (done_when 4): docs/audits/3750-whole-file-model-reads.md rows updated; after this PR the same git grep finds 215 hits (231 - 17 + 1: aprender-serve read_magic''s take(4) before read_to_end matches the pattern). CHECKS: cargo check -p apr-cli --features cuda --lib --tests rc 0 (the bench''s CUDA path takes the map); clippy -p apr-format / aprender-core / aprender-serve / apr-cli --lib -D warnings rc 0; clippy --lib --tests reports nothing in this PR''s files (aprender-serve''s test targets carry pre-existing errors in iq2_s.rs and falsification_crux_c_34.rs, the same on origin/main). NOT HERE: apr qa''s tensor_contract gate still reads the whole GGUF onto the heap (GgufReader::from_file). That is WHOLE-DATA in the ledger, 19.36 GB RssAnon on the 30B, filed as #3790. ALSO HERE (aprender-91, 3fafc55915, #4520 step 2): apr inspect on a GGUF was still whole-file: RosettaStone::inspect_gguf read the model and copied every tensor. It now parses the header from a growing prefix (GgufReader::header_from_file, 8 MiB first, doubling) and sizes tensors from the header (tensor_extents, bounds-checked against the file length). MEASURED on Qwen3.5-27B-Q4_K_M.gguf (16,740,812,704 B), apr inspect --json, rc 0.70.0 817d63361 vs this branch: bytes read 16,740,812,720 -> 16,777,232; peak RSS 32.8 GB -> 75 MB; wall 62.1 s -> 0.24 s; JSON byte-identical. Case row: prefix_rss_tests peak_rss_probe runs inspect on the sparse 2 GiB GGUF; with the whole-file read put back, 2,124,048 KiB vs the 262,144 KiB bound (RED).' diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 06cc47fb2e..83ae140615 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -21020,6 +21020,28 @@ roadmap: labels: - kind:code notes: 'SCOPE (issue title, amended by the cop from a8''s trace): SafeTensors run and every AprTransformer caller never sampled. `apr run` on .apr already sampled (run_apr_quantized_cpu_inference -> OwnedQuantizedModel::generate_with_cache, the GGUF sampler), which is measured below. ONE SHARED SAMPLER (done_when 1): new `realizar::sampling` with `draw(logits, temperature, top_k, top_p, r)` (the body moved verbatim from OwnedQuantizedModel::sample_topk_with_draw, which now delegates), `draw_seeded(.., &mut StdRng)`, `is_greedy(temperature, top_k)` (temperature 0 or top_k 1; top_k 0 disables the filter) and `DEFAULT_SEED` (42). Greedy stays each loop''s OWN argmax because the loops break ties differently (max_by keeps the last maximum, gguf ops::argmax the first), so greedy output is byte-identical. Routed through it: (a) AprTransformer `sample_from_logits` (apr_transformer/generation.rs), which took the ARGMAX of the top-k/top-p survivors and so could never draw. It is now seeded from a new `GenerateConfig.seed`, and `top_k_top_p_survivors` is deleted (the filter lives in the shared draw, pinned by the GGUF top-k/top-p rows). (b) The SafeTensors `apr run` CPU loop, `greedy_decode_with_transformer` -> `decode_with_transformer(transformer, input, &InferenceConfig)` (plain and sharded). It never read temperature/top_k/top_p/seed. (c) GpuModel `ops::sample_topk` and `kv_forward_block::sample_topk`: both sorted by probability and returned the first entry, the argmax. They are now seeded from a new `GpuGenerateConfig.seed` (chat takes the OpenAI `seed`). GREEDY-ONLY DECODERS: a sampled request no longer silently takes them. The SafeTensors CUDA `generate(input, max_tokens, eos_id)` and both wgpu loops (GGUF `try_wgpu_generate`, `try_apr_wgpu_inference`, each an inline argmax) run only when `is_greedy`; otherwise a notice is printed and the CPU loop that draws runs. The GpuModel `/v1/completions` handler hardcoded `top_k: 1`, so a request temperature did nothing; it now uses the CPU completions handlers'' rule (1 at temperature 0, else 40). #3754''s DEFAULT_TOP_K replaces that literal when both land. DEFAULT CHANGE: `apr_transformer::GenerateConfig::default()` temperature 1.0 -> 0.0. With the old argmax sampler every config, this default included, decoded greedily, so default callers (bench, serve /generate with no temperature) keep their output. A default of 1.0 with a real draw would have randomised them. Five rows that pinned 1.0 were updated with that reason. EXISTING ROWS THAT PINNED THE DEFECT: PMAT-820''s `neutral_params_byte_identical_to_legacy` and `top_k_ge_vocab_is_neutral` asserted that sampling at T>0 returned the argmax. They are replaced by sampler_tests (draws off the argmax; same seed reproduces, another seed changes; T0/top-k 1 is the argmax; top-k 2 bounds the draw). TESTS: sampling::tests (is_greedy table, the CDF walk, seeded reproducibility); apr_transformer::generation::sampler_tests (4); infer::tests_sampling_3760 through `run_inference` on the tiny SafeTensors fixture (sampled != greedy for one of 8 seeds, same seed identical and another seed differs, top-k 1 / T0 greedy on 4 seeds) plus `wgpu_serves_greedy_requests_only`; gpu::scheduler test_sample_topk_draws_off_the_argmax_and_is_seeded. MUTANTS, each run and each RED: AprTransformer sampler back to argmax (3 red); SafeTensors loop always greedy (2); SafeTensors seed ignored (1); GpuModel kv sampler back to argmax (1). MEASURED (done_when 2) on lambda, CPU, qwen2.5-coder-0.5b-instruct .apr and .safetensors, pinned via scripts/apr_bin.sh, 24 tokens. main `apr 0.69.0 (a9502d992)`: .safetensors sampled != greedy NO, seed 99 != 1234 NO (the defect); .apr YES/YES. Branch `(8760145f5)`: both formats sampled != greedy YES, same seed x2 IDENTICAL, seed 99 != 1234 YES, --top-k 1 == greedy YES, and greedy tokens byte-identical to main on both. wgpu: without --no-gpu on lambda (Qwen3-1.7B) main tries wgpu (Vulkan), its parity gate rejects it (cosine 0.870116), and it falls back to CPU. So main already sampled there, and the greedy-wgpu case is traced from code, not measured. On the branch a sampled request prints the notice and goes straight to the CPU. SWEEP (done_when 3): docs/audits/sampler-sweep-PMAT-3760.md lists every site in 4 sections: A routed through the shared sampler, B greedy-only decoders gated, C greedy by construction, D draws but not via the shared seeded draw, each with a named follow-up; the highest-priority follow-up is `sample_with_temperature` (wall-clock hash RNG) on serve''s APR Q4K GPU chat. SUITES at 8760145f5: aprender-serve lib 15920 passed; apr-cli lib 7300 passed; aprender-contracts lib 1689 passed; aprender-serve integration apr_coverage/apr_transformer_coverage/apr_transformer_deep_coverage 258+50+90 passed, and gguf_model_coverage, gguf_transformer_coverage, gpu_coverage, mock_gpu_flow and moe_kv_cache_equivalence 190+38+306+30+0(1 ignored) passed; clippy -D warnings clean on aprender-serve, apr-cli and aprender-serve --features cuda; fmt clean. `cargo check -p aprender-serve --features cuda --tests` fails ONLY on tests/gpu_cpu_trace_compare.rs (`missing field lm_head_tied`). That is pre-existing: this branch does not touch that file or AprTransformer, and gpu/tests/imp_1001d.rs notes that no CI job builds the cuda test profile.' +- id: PMAT-3761 + github_issue: 3761 + item_type: task + title: '#3750 PR B: the 17 header-only whole-file model reads outside apr qa — magic/header/metadata readers in their format crates' + status: in_progress + priority: critical + assigned_to: aprender-0e + created: 2026-09-21T21:52:15Z + updated: 2026-09-21T21:52:15Z + spec: null + acceptance_criteria: + - 'THIS PR IS #3750 PR B (#3761, 0.69.1). Cop ruling (aprender-04 to aprender-0e, 2026-09-21), verbatim: "PR B = a sub-issue in **0.69.1**, not later." Stacked on #3750 PR A (PMAT-3750), whose ledger this PR updates.' + - 'Issue #3761 done_when 1, verbatim: "each site above reads only what it uses (a bounded magic read, or the format''s header prefix); the APR v2 prefix reader lives in apr-format and the SafeTensors one in aprender-core, sharing ONE bounded-prefix policy (first read + cap) with apr-cli''s model_header.rs"' + - 'Issue #3761 done_when 2, verbatim: "the `list_tensors` GGUF/APR v1 path reads tensor data only when `--stats` asks for it"' + - 'Issue #3761 done_when 3, verbatim: "a case row per reader proving it never touches tensor data (a sparse multi-GB fixture + a peak-RSS bound, as PR A''s), and a mutant that puts a whole-file read back goes RED"' + - 'Issue #3761 done_when 4, verbatim: "the ledger''s rows are updated to CONVERTED"' + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'Closes #3761; Refs #3750. WHAT LANDS. ONE bounded-prefix policy: crates/apr-format/src/prefix.rs (HEADER_FIRST_READ 16 MiB, doubling, HEADER_READ_CAP 256 MiB, refused past it, never read whole; read_prefix, parse_growing_prefix, apr_v2_header_prefix bounded by the header''s data_offset). crates/aprender-core/src/format/prefix.rs re-exports it and adds safetensors_header_prefix (8-byte length + JSON header, capped) beside the SafeTensors format, as done_when 1 places it. apr-cli model_header.rs delegates to it and adds gguf_header_bytes (the header cut at data_offset, for LlamaTokenizer::from_gguf_bytes, which is not prefix-safe). The 17 sites (ledger rows now CONVERTED (#3761), each naming what it reads now and its case row): bench format detect 8 bytes; bench GGUF maps ONCE and hands the map to the CPU, CUDA and MoE paths (each mapped again before; realizar''s map pre-faults every page, MAP_POPULATE); gguf_vocab via gguf_header_bytes; eval count_safetensors_keys via safetensors_header_prefix; verify_single_file reads the header and STREAMS its FNV-1a hash in 1 MiB chunks (a test proves the streamed hash equals the whole-buffer hash); the converter''s five APR readers via apr_v2_header_prefix or a read bounded by the metadata section; lint SafeTensors (length + header) and APR (apr_v2_header_prefix); rosetta inspect_apr; is_onnx_file 5 bytes; aprender-serve is_apr_file / format_from_magic 4 bytes (read_magic). list_tensors (done_when 2): GGUF without --stats parses the header from a growing prefix; the old path also COPIED the bytes (list_tensors_gguf data.to_vec()), 4.2 GB peak on the 2 GiB fixture. GGUF with --stats and APR v1 use a map, which pages in only what is read. extract_apr_tokenizer_hint: APR v2 input is scanned within apr_v2_header_prefix. The v2 writer zero-pads its metadata, so the terminator this legacy scan looks for is absent and the answer is None, as it was when it read the whole model. Any other layout gets a growing prefix under the cap, and test_infer_tokenizer_json_legacy_apr_with_tokenizer is its positive control. CASE ROWS (done_when 3): each is a real file extended SPARSELY to 2 GiB, the readers run in a CHILD process (the test binary re-run on one probe test), and VmHWM is held under 256 MiB. apr-format 4,608 KiB; aprender-core format row 36.1-37.1 MiB and converter row 15.8-17.2 MiB; aprender-serve 19.9-20.3 MiB; apr-cli model_header 42.4-43.8 MiB, gguf_vocab 39.8-42.3 MiB, eval 25.3-27.3 MiB. apr bench''s bound is one map + 256 MiB, because the bench runs the model: 2,123,680-2,125,672 KiB. MUTANTS (done_when 3): a whole-file read put back into each reader, 22 in all (apr-format 1; aprender-core 12, covering list_tensors GGUF and APR v1, lint x2, inspect, is_onnx_file, safetensors_header_prefix and the converter''s five; aprender-serve 2; apr-cli 7, covering gguf_header_bytes, gguf_vocab, count_safetensors_keys, verify_single_file header and hash, bench format detect and bench tokenizer read). Every one goes RED at 2,100,864-4,218,352 KiB. Each probe reports its peak before its result checks, so a mutant fails on RSS, not on a side assertion. Ledger (done_when 4): docs/audits/3750-whole-file-model-reads.md rows updated; after this PR the same git grep finds 215 hits (231 - 17 + 1: aprender-serve read_magic''s take(4) before read_to_end matches the pattern). CHECKS: cargo check -p apr-cli --features cuda --lib --tests rc 0 (the bench''s CUDA path takes the map); clippy -p apr-format / aprender-core / aprender-serve / apr-cli --lib -D warnings rc 0; clippy --lib --tests reports nothing in this PR''s files (aprender-serve''s test targets carry pre-existing errors in iq2_s.rs and falsification_crux_c_34.rs, the same on origin/main). NOT HERE: apr qa''s tensor_contract gate still reads the whole GGUF onto the heap (GgufReader::from_file). That is WHOLE-DATA in the ledger, 19.36 GB RssAnon on the 30B, filed as #3790. ALSO HERE (aprender-91, 3fafc55915, #4520 step 2): apr inspect on a GGUF was still whole-file: RosettaStone::inspect_gguf read the model and copied every tensor. It now parses the header from a growing prefix (GgufReader::header_from_file, 8 MiB first, doubling) and sizes tensors from the header (tensor_extents, bounds-checked against the file length). MEASURED on Qwen3.5-27B-Q4_K_M.gguf (16,740,812,704 B), apr inspect --json, rc 0.70.0 817d63361 vs this branch: bytes read 16,740,812,720 -> 16,777,232; peak RSS 32.8 GB -> 75 MB; wall 62.1 s -> 0.24 s; JSON byte-identical. Case row: prefix_rss_tests peak_rss_probe runs inspect on the sparse 2 GiB GGUF; with the whole-file read put back, 2,124,048 KiB vs the 262,144 KiB bound (RED).' - id: PMAT-3770 github_issue: 3770 item_type: task