Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 25 additions & 0 deletions .github/workflows/cuda-nightly.yml
Original file line number Diff line number Diff line change
Expand Up @@ -688,6 +688,31 @@ jobs:
fi
echo "#4250 G0: GREEN passed, per-token RED control failed as required (rc=$r)"

# #4523: the qwen35 CPU/GPU parity module on sm_121. ci.yml's cuda-unit runs it
# on yoga-eph with no models, so every test there SKIPs as model-absent: the
# whole-forward budgets were only ever read on sm_89 + x86_64, and GB10 +
# aarch64 sat 7.105e-2 over a 7e-2 budget with no lane to say so. This lane
# runs the module where the model is and refuses a skip or an empty selection.
- name: qwen35 CUDA parity module on sm_121 (aprender#4523)
if: ${{ !cancelled() && steps.decide.outputs.proceed == 'true' }}
run: |
set -uo pipefail
m=gguf::cuda::forward_qwen35_cuda::qwen35_cuda_tests::
model="$HOME/models/Qwen3.5-0.8B-Q4_K_M.gguf"
[ -f "$model" ] || { echo "::error::#4523 UNMEASURABLE on ${{ matrix.name }} (provisioning): $model is absent"; exit 1; }
nice -n 19 cargo test -p aprender-serve --features cuda --lib --release -- \
"$m" --test-threads 1 --nocapture > q35-parity.log 2>&1; rc=$?
grep -E '^\[e2e\]|^test result' q35-parity.log | tail -n 20
if [ "$rc" -ne 0 ]; then echo "::error::#4523 qwen35 parity module failed on sm_121 (rc=$rc)"; exit 1; fi
if grep -qE 'is absent|SKIP' q35-parity.log; then
grep -E 'is absent|SKIP' q35-parity.log | head -5
echo "::error::#4523 a qwen35 parity test skipped on the host that holds the model"; exit 1
fi
# The libtest summary, not '... ok' lines: --nocapture output can split those.
n=$(sed -nE 's/^test result: ok\. ([0-9]+) passed.*/\1/p' q35-parity.log | tail -1)
if [ "${n:-0}" -eq 0 ]; then echo "::error::#4523 the module filter selected ZERO tests"; exit 1; fi
echo "#4523: ${n} qwen35 parity test(s) passed on sm_121, 0 skips"

# ── ada-yoga: the sm_89 leg, restored ─────────────────────────────────────
#
# RESTORING A DELETION, NOT INVENTING A LANE. #2740 removed the x86_64 leg
Expand Down
49 changes: 45 additions & 4 deletions crates/aprender-serve/src/gguf/cuda/forward_qwen35_cuda_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -111,11 +111,33 @@ const KV_TOL: f32 = TOL;
/// **-0.246**, so the assertion discriminates rather than decorates.
const COSINE_FLOOR: f32 = 0.996;

/// The measured end-to-end logit L∞ — 6.721e-2 relative, at position 0 —
/// rounded UP to one significant figure. A budget on the accumulated
/// CPU-reference activation quantization through 24 layers and the `lm_head`,
/// not a tolerance anyone should read as accuracy: see [`COSINE_FLOOR`].
/// The measured end-to-end logit L∞, rounded UP to one significant figure. A
/// budget on the accumulated CPU-reference activation quantization through 24
/// layers and the `lm_head`, not a tolerance anyone should read as accuracy:
/// see [`COSINE_FLOOR`].
///
/// It is keyed on the HOST CPU architecture, not the GPU, because the reference
/// is what moves (aprender#4523). Measured at position 0 on the same GGUF
/// (sha256 `bd258782e35f7f45…`), dumps diffed across hosts:
///
/// | pair | relative L∞ | cosine |
/// |---|---|---|
/// | GPU sm_89 (4090) vs GPU sm_121 (GB10) | **6.5e-7** | 1.000000 |
/// | CPU x86_64 AVX2 vs CPU aarch64 NEON | **2.6e-2** | 0.999743 |
/// | GPU vs CPU, x86_64 host | 6.721e-2 | 0.998348 |
/// | GPU vs CPU, aarch64 host | 7.105e-2 | 0.998316 |
///
/// The two GPUs agree to f32 rounding; the two CPU references do not, because
/// their Q8_K activation-quant dot products take different SIMD paths. So the
/// budget follows the reference: x86_64 worst 6.721e-2 -> **7e-2**; aarch64
/// worst 7.105e-2 (positions 1..5: 2.226e-2, 3.643e-2, 2.659e-2, 2.524e-2,
/// 2.158e-2) -> **8e-2**. Re-derive with `QWEN35_E2E_DUMP=<dir>`, which prints
/// every position's reading and then fails, so it can never pass as a verdict.
#[cfg(not(target_arch = "aarch64"))]
const LOGITS_BUDGET: f32 = 7e-2;
/// See the x86_64 [`LOGITS_BUDGET`]: the aarch64 CPU reference, measured on GB10.
#[cfg(target_arch = "aarch64")]
const LOGITS_BUDGET: f32 = 8e-2;

/// The same, for one whole attention layer's output hidden state: measured
/// 4.867e-2 relative (layer 15, position 0), rounded up to one significant
Expand Down Expand Up @@ -1035,6 +1057,7 @@ fn qwen35_cuda_forward_single_matches_cpu_logits_end_to_end() {
let mut cpu_state = qwen.new_state(LONG_PROMPT.len() + 1);
let mut worst_cos = 1.0f32;
let mut worst_linf = 0.0f32;
let mut dumped = false;

for (pos, &token) in LONG_PROMPT.iter().enumerate() {
let want = qwen
Expand All @@ -1048,6 +1071,19 @@ fn qwen35_cuda_forward_single_matches_cpu_logits_end_to_end() {
.forward_single(token, &mut gpu_state, pos)
.expect("gpu forward");
let gpu_ms = t0.elapsed().as_secs_f64() * 1e3;
if let Ok(dir) = std::env::var("QWEN35_E2E_DUMP") {
for (side, v) in [("gpu", &got), ("cpu", &want)] {
let bytes: Vec<u8> = v.iter().flat_map(|x| x.to_le_bytes()).collect();
std::fs::write(format!("{dir}/{side}_pos{pos}.f32"), bytes).expect("dump");
}
eprintln!(
"[e2e-dump] pos {pos}: cosine {:.6} relative L-inf {:.3e}",
cosine(&got, &want),
rel_linf(&got, &want)
);
dumped = true;
continue;
}
let (cos, linf) =
assert_forward_parity(&got, &want, LOGITS_BUDGET, &format!("pos {pos} logits"));
eprintln!(
Expand All @@ -1063,6 +1099,11 @@ fn qwen35_cuda_forward_single_matches_cpu_logits_end_to_end() {
"pos {pos}: the device KV cache must have advanced"
);
}
assert!(
!dumped,
"QWEN35_E2E_DUMP is a measurement run: the readings are printed above and nothing \
was asserted, so it must not read as a pass"
);
eprintln!(
"[e2e] worst over {} positions: cosine {worst_cos:.6} relative L-inf {worst_linf:.3e}",
LONG_PROMPT.len(),
Expand Down
Loading
Loading