Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions crates/apr-cli/src/commands/bench.rs
Original file line number Diff line number Diff line change
Expand Up @@ -828,6 +828,10 @@ fn print_header(path: &Path, config: &BenchConfig) {
}

include!("benchmark.rs");

#[cfg(all(test, feature = "inference", target_os = "linux"))]
#[path = "bench_rss_tests.rs"]
mod rss_tests;
include!("bench_safetensors.rs");
include!("bench_moe.rs");
include!("bench_qwen35.rs");
Expand Down
11 changes: 4 additions & 7 deletions crates/apr-cli/src/commands/bench_moe.rs
Original file line number Diff line number Diff line change
Expand Up @@ -31,24 +31,21 @@ fn is_moe_gguf(gguf: &realizar::gguf::GGUFModel) -> bool {
/// then runs warmup + iterations through the appropriate forward path.
#[cfg(feature = "inference")]
fn run_gguf_moe_benchmark(
path: &Path,
mapped: &realizar::gguf::MappedGGUFModel,
config: &BenchConfig,
use_cuda: bool,
prompt_tokens: &[u32],
start: Instant,
_tracer: &TracerImpl,
) -> Result<BenchResult> {
use realizar::gguf::qwen3_moe_load::load_qwen3_moe_layer;
use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel};
use realizar::gguf::OwnedQuantizedModel;

if !config.quiet {
eprintln!("{}", "Loading MoE GGUF model...".yellow());
}
let start = Instant::now();

let mapped = MappedGGUFModel::from_path(path)
.map_err(|e| CliError::ValidationFailed(format!("Failed to mmap MoE model: {e}")))?;

let model = OwnedQuantizedModel::from_mapped(&mapped)
let model = OwnedQuantizedModel::from_mapped(mapped)
.map_err(|e| CliError::ValidationFailed(format!("Failed to create MoE model: {e}")))?;

// Read MoE config from GGUF metadata. expert_count() must be Some
Expand Down
51 changes: 51 additions & 0 deletions crates/apr-cli/src/commands/bench_rss_tests.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
//! #3761 case row: `apr bench` on a GGUF detects the format from 8 bytes and maps the model
//! ONCE. The bench runs the model, so the map is its floor: realizar's map pre-faults every page
//! (MAP_POPULATE, PMAT-304), which puts the whole file in RSS. The row proves nothing is read on
//! top of that one map: it read the whole file onto the heap first, and each backend path then
//! mapped it again. The fixture's header parses and its model does not build (one tensor), so
//! the run stops after the reads under test.
//!
//! Measured (x86-64 debug, 2026-09-21): 2,123,680-2,125,672 KiB over three runs, i.e. the
//! 2,097,152 KiB map and about 27 MiB. The whole-file read put back into the format check, or
//! into the tokenizer's parse, gives 4,212,352 / 4,212,152 KiB, and so did the second map this
//! change removed (4,224,696 KiB).

use super::*;
use crate::commands::model_header::rss_probe;

#[test]
fn bench_of_a_2_gib_gguf_keeps_peak_rss_small() {
let f = rss_probe::sparse(
&rss_probe::gguf_with_vocab(),
rss_probe::FIXTURE_LEN,
".gguf",
);
let hwm = rss_probe::child_peak_kb(
"commands::bench::rss_tests::bench_peak_rss_probe",
&[(BENCH_PROBE, f.path())],
);
// One map of the file, plus the margin every header row gets.
let bound = rss_probe::FIXTURE_LEN / 1024 + rss_probe::PEAK_RSS_BOUND_KB;
assert!(
hwm < bound,
"peak RSS {hwm} KiB benchmarking a 2 GiB GGUF (bound {bound} KiB, one map + margin): \
a whole-file read or a second map is back"
);
}

const BENCH_PROBE: &str = "APR_3761_BENCH_PROBE";

/// Not a test on its own: with the probe variable unset it does nothing.
#[test]
fn bench_peak_rss_probe() {
let Some(path) = std::env::var_os(BENCH_PROBE) else {
return;
};
let config = BenchConfig {
quiet: true,
..BenchConfig::default()
};
let result = run_realizar_benchmark(Path::new(&path), &config);
assert!(result.is_err(), "a one-tensor GGUF cannot be benchmarked");
rss_probe::report_peak();
}
19 changes: 6 additions & 13 deletions crates/apr-cli/src/commands/bench_safetensors.rs
Original file line number Diff line number Diff line change
Expand Up @@ -273,25 +273,20 @@ fn run_cuda_measurement(
#[cfg_attr(coverage_nightly, coverage(off))]
#[cfg(all(feature = "inference", feature = "cuda"))]
fn run_cuda_benchmark(
_gguf: &realizar::gguf::GGUFModel,
_model_bytes: &[u8],
mapped: &realizar::gguf::MappedGGUFModel,
prompt_tokens: &[u32],
gen_config: &realizar::gguf::QuantizedGenerateConfig,
config: &BenchConfig,
start: Instant,
model_path: &Path,
tracer: &TracerImpl,
) -> Result<BenchResult> {
use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel, OwnedQuantizedModelCuda};
use realizar::gguf::{OwnedQuantizedModel, OwnedQuantizedModelCuda};

if !config.quiet {
eprintln!("{}", "Initializing CUDA model...".cyan());
}

let mapped = MappedGGUFModel::from_path(model_path)
.map_err(|e| CliError::ValidationFailed(format!("Failed to map model: {e}")))?;

let model = OwnedQuantizedModel::from_mapped(&mapped)
let model = OwnedQuantizedModel::from_mapped(mapped)
.map_err(|e| CliError::ValidationFailed(format!("Failed to create model: {e}")))?;

let mut cuda_model = OwnedQuantizedModelCuda::new(model, 0)
Expand All @@ -318,18 +313,16 @@ fn run_cuda_benchmark(
/// CPU-based benchmark fallback path
#[cfg(feature = "inference")]
fn run_cpu_benchmark(
mapped: &realizar::gguf::MappedGGUFModel,
prompt_tokens: &[u32],
gen_config: &realizar::gguf::QuantizedGenerateConfig,
config: &BenchConfig,
start: Instant,
path: &Path,
tracer: &TracerImpl,
) -> Result<BenchResult> {
use realizar::gguf::{MappedGGUFModel, OwnedQuantizedModel};
use realizar::gguf::OwnedQuantizedModel;

let mapped = MappedGGUFModel::from_path(path)
.map_err(|e| CliError::ValidationFailed(format!("Failed to mmap model: {e}")))?;
let model = OwnedQuantizedModel::from_mapped(&mapped)
let model = OwnedQuantizedModel::from_mapped(mapped)
.map_err(|e| CliError::ValidationFailed(format!("Failed to create model: {e}")))?;

bench_log_ready(config, start.elapsed(), " (CPU)");
Expand Down
36 changes: 14 additions & 22 deletions crates/apr-cli/src/commands/benchmark.rs
Original file line number Diff line number Diff line change
Expand Up @@ -100,12 +100,12 @@ fn print_results(result: &BenchResult) {
fn run_realizar_benchmark(path: &Path, config: &BenchConfig) -> Result<BenchResult> {
use realizar::format::{detect_format, ModelFormat};

// Read first 8 bytes for format detection
let header_bytes = std::fs::read(path)
// Read first 8 bytes for format detection (#3761: it read the whole model under this comment)
let header_bytes = super::model_header::read_prefix(path, 8)
.map_err(|e| CliError::ValidationFailed(format!("Failed to read model: {e}")))?;

// Detect format
let format = detect_format(&header_bytes[..8.min(header_bytes.len())])
let format = detect_format(&header_bytes)
.map_err(|e| CliError::ValidationFailed(format!("Failed to detect format: {e}")))?;

if !config.quiet {
Expand Down Expand Up @@ -156,18 +156,19 @@ fn run_gguf_benchmark(
use_cuda: bool,
tracer: &TracerImpl,
) -> Result<BenchResult> {
use realizar::gguf::{GGUFModel, QuantizedGenerateConfig};
use realizar::gguf::{MappedGGUFModel, QuantizedGenerateConfig};

if !config.quiet {
eprintln!("{}", "Loading GGUF model...".yellow());
}
let start = Instant::now();

// Load model for tokenization
let model_bytes = std::fs::read(path)
.map_err(|e| CliError::ValidationFailed(format!("Failed to read model: {e}")))?;
let gguf = GGUFModel::from_bytes(&model_bytes)
.map_err(|e| CliError::ValidationFailed(format!("Failed to parse GGUF: {e}")))?;
// The ONE map of the model (#3761). The tokenizer and the MoE fields come from its header,
// and inference runs on it. This read the whole file first, and each path then mapped it
// again: a map pre-faults every page (MAP_POPULATE), so two maps doubled the peak.
let mapped = MappedGGUFModel::from_path(path)
.map_err(|e| CliError::ValidationFailed(format!("Failed to map GGUF: {e}")))?;
let gguf = &mapped.model;

// Tokenize prompt
let bos = aprender::demo::SpecialTokens::qwen2().bos_id;
Expand Down Expand Up @@ -212,7 +213,7 @@ fn run_gguf_benchmark(
gguf.expert_used_count().unwrap_or(0)
);
}
return run_gguf_moe_benchmark(path, config, use_cuda, &prompt_tokens, tracer);
return run_gguf_moe_benchmark(&mapped, config, use_cuda, &prompt_tokens, start, tracer);
}

let gen_config = QuantizedGenerateConfig {
Expand All @@ -224,16 +225,7 @@ fn run_gguf_benchmark(

#[cfg(feature = "cuda")]
if use_cuda {
match run_cuda_benchmark(
&gguf,
&model_bytes,
&prompt_tokens,
&gen_config,
config,
start,
path,
tracer,
) {
match run_cuda_benchmark(&mapped, &prompt_tokens, &gen_config, config, start, tracer) {
Ok(result) => return Ok(result),
Err(e) => {
// GH-284: Fall back to CPU on CUDA capability mismatch (e.g. missing QkNorm kernel)
Expand All @@ -245,17 +237,17 @@ fn run_gguf_benchmark(
}
let cpu_start = Instant::now();
return run_cpu_benchmark(
&mapped,
&prompt_tokens,
&gen_config,
config,
cpu_start,
path,
tracer,
);
}
}
}
run_cpu_benchmark(&prompt_tokens, &gen_config, config, start, path, tracer)
run_cpu_benchmark(&mapped, &prompt_tokens, &gen_config, config, start, tracer)
}

/// Resolve prompt tokens from APR model's tokenizer, with fallbacks.
Expand Down
5 changes: 3 additions & 2 deletions crates/apr-cli/src/commands/embed_viz.rs
Original file line number Diff line number Diff line change
Expand Up @@ -379,10 +379,11 @@ fn resolve_tokens(
/// read ONCE and can also ask it how many tokens the model declares — the
/// cross-check in `check_vocab_axis`.
fn gguf_vocab(model: &Path) -> Option<LlamaTokenizer> {
let bytes = std::fs::read(model).ok()?;
if !bytes.starts_with(b"GGUF") {
// #3761: the complete header only (the vocabulary lives there), never the tensor data
if super::model_header::read_prefix(model, 4).ok()?.as_slice() != b"GGUF" {
return None;
}
let bytes = super::model_header::gguf_header_bytes(model).ok()?;
LlamaTokenizer::from_gguf_bytes(&bytes).ok()
}

Expand Down
39 changes: 39 additions & 0 deletions crates/apr-cli/src/commands/embed_viz_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -595,3 +595,42 @@ fn pca_needs_at_least_two_rows() {
let err = run(&a).expect_err("PCA on one sample has no variance to decompose");
assert!(err.to_string().contains("at least 2 rows"), "got: {err}");
}

/// #3761 case row: `gguf_vocab` reads a 2 GiB GGUF's header (where the vocabulary lives) and
/// never its tensor data. Measured (x86-64 debug, 2026-09-21): 39.8-42.3 MiB over three runs; with `gguf_vocab` reading the
/// whole file again, 2,116,644 KiB, and with `gguf_header_bytes` doing so, 2,117,924 KiB.
#[cfg(target_os = "linux")]
#[test]
fn gguf_vocab_of_a_2_gib_gguf_keeps_peak_rss_small() {
use crate::commands::model_header::rss_probe;
let f = rss_probe::sparse(
&rss_probe::gguf_with_vocab(),
rss_probe::FIXTURE_LEN,
".gguf",
);
let hwm = rss_probe::child_peak_kb(
"commands::embed_viz::tests::gguf_vocab_peak_rss_probe",
&[(VOCAB_PROBE, f.path())],
);
assert!(
hwm < rss_probe::PEAK_RSS_BOUND_KB,
"peak RSS {hwm} KiB reading the vocabulary of a 2 GiB GGUF (bound {} KiB): a whole-file read is back",
rss_probe::PEAK_RSS_BOUND_KB
);
}

#[cfg(target_os = "linux")]
const VOCAB_PROBE: &str = "APR_3761_VOCAB_PROBE";

/// Not a test on its own: with the probe variable unset it does nothing.
#[cfg(target_os = "linux")]
#[test]
fn gguf_vocab_peak_rss_probe() {
let Some(path) = std::env::var_os(VOCAB_PROBE) else {
return;
};
let tokenizer =
gguf_vocab(std::path::Path::new(&path)).expect("the vocabulary, from the header");
assert_eq!(tokenizer.vocab_size(), 4);
crate::commands::model_header::rss_probe::report_peak();
}
74 changes: 74 additions & 0 deletions crates/apr-cli/src/commands/eval/eval_mod_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,26 @@ fn format_archive_size_units() {
assert_eq!(format_archive_size(1_610_612_736), "1.5 GB");
}

// ── compute_file_hash_streamed (#3761): the same hash, chunk by chunk ──────

#[test]
fn streamed_file_hash_equals_the_whole_buffer_hash() {
use std::io::Write;
// 2.5 MiB, so the 1 MiB chunks split it three ways, one part partial
let bytes: Vec<u8> = (0..(5usize << 19)).map(|i| (i * 31 % 251) as u8).collect();
let mut f = tempfile::NamedTempFile::new().expect("temp file");
f.write_all(&bytes).expect("write");
assert_eq!(
compute_file_hash_streamed(f.path()).expect("stream the hash"),
compute_file_hash(&bytes)
);
let empty = tempfile::NamedTempFile::new().expect("temp file");
assert_eq!(
compute_file_hash_streamed(empty.path()).expect("empty"),
"cbf29ce484222325"
);
}

// ── compute_file_hash (FNV-1a) ─────────────────────────────────────────────

#[test]
Expand Down Expand Up @@ -479,3 +499,57 @@ fn default_prompts_are_nonempty_python() {
assert!(p.contains("def ") || p.contains("class "));
}
}

/// #3761 case row: `count_safetensors_keys` reads a 2 GiB SafeTensors file's JSON header, and
/// `verify_single_file` hashes every byte of it STREAMED; neither holds the file in memory.
/// Measured (x86-64 debug, 2026-09-21): 25.3-27.3 MiB over three runs. A whole-file read put back
/// into `count_safetensors_keys`, into `verify_single_file`'s header read, or into its hash:
/// 2,115,124 / 2,114,024 / 2,114,080 KiB.
#[cfg(target_os = "linux")]
#[test]
fn safetensors_readers_of_a_2_gib_file_keep_peak_rss_small() {
use crate::commands::model_header::rss_probe;
let f = rss_probe::sparse(
&rss_probe::safetensors(),
rss_probe::FIXTURE_LEN,
".safetensors",
);
let hwm = rss_probe::child_peak_kb(
"commands::eval::eval_mod_tests::safetensors_readers_peak_rss_probe",
&[(EVAL_PROBE, f.path())],
);
assert!(
hwm < rss_probe::PEAK_RSS_BOUND_KB,
"peak RSS {hwm} KiB on a 2 GiB SafeTensors file (bound {} KiB): a whole-file read is back",
rss_probe::PEAK_RSS_BOUND_KB
);
}

#[cfg(target_os = "linux")]
const EVAL_PROBE: &str = "APR_3761_EVAL_PROBE";

/// Not a test on its own: with the probe variable unset it does nothing.
#[cfg(target_os = "linux")]
#[test]
fn safetensors_readers_peak_rss_probe() {
let Some(path) = std::env::var_os(EVAL_PROBE) else {
return;
};
let path = std::path::Path::new(&path);
assert_eq!(count_safetensors_keys(path), 1);
let mut checks = Vec::new();
verify_single_file(path, &mut checks).expect("verify");
assert!(
checks
.iter()
.any(|(name, ok)| name == "1 tensors found" && *ok),
"{checks:?}"
);
assert!(
checks
.iter()
.any(|(name, _)| name.starts_with("BLAKE3 hash: ")),
"{checks:?}"
);
crate::commands::model_header::rss_probe::report_peak();
}
Loading
Loading