diff --git a/crates/apr-cli/src/commands/serve/chat.rs b/crates/apr-cli/src/commands/serve/chat.rs index ef240b175d..b0ff459b02 100644 --- a/crates/apr-cli/src/commands/serve/chat.rs +++ b/crates/apr-cli/src/commands/serve/chat.rs @@ -266,7 +266,7 @@ pub(crate) async fn safetensors_chat_completions_handler( .get("temperature") .and_then(|t| t.as_f64()) .unwrap_or(0.0) as f32; - let output_ids = { + let (output_ids, max_tokens) = { // PMAT-189: Handle transformer lock poisoning gracefully let t = match transformer.lock() { Ok(guard) => guard, @@ -280,8 +280,12 @@ pub(crate) async fn safetensors_chat_completions_handler( .into_response(); } }; - match st_cpu_generate(&t, &input_ids, max_tokens, temperature) { - Ok(ids) => ids, + let budget = match st_context_budget(&t, input_ids.len(), max_tokens) { + Ok(budget) => budget, + Err(refusal) => return refusal, + }; + match st_cpu_generate(&t, &input_ids, budget, temperature) { + Ok(ids) => (ids, budget), Err(e) => { return ( StatusCode::INTERNAL_SERVER_ERROR, @@ -336,6 +340,7 @@ pub(crate) async fn safetensors_chat_completions_handler( stream_mode, input_ids.len(), tokens_generated, + max_tokens, elapsed, tok_per_sec, ) @@ -419,6 +424,7 @@ fn build_chat_response( stream_mode: bool, prompt_tokens: usize, tokens_generated: usize, + max_tokens: usize, elapsed: std::time::Duration, tok_per_sec: f64, ) -> axum::response::Response { @@ -427,7 +433,12 @@ fn build_chat_response( let request_id = generate_request_id(); let has_tool_calls = tool_calls.is_some(); - let finish_reason = if has_tool_calls { "tool_calls" } else { "stop" }; + // #3718: a reply cut at `max_tokens` is "length", never "stop". + let finish_reason = if has_tool_calls { + "tool_calls" + } else { + super::handlers::finish_reason_for(tokens_generated, max_tokens) + }; let created = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .unwrap_or_default() @@ -688,6 +699,26 @@ mod chat_helper_tests { // ---- build_chat_response ------------------------------------------ + /// #3718: a SafeTensors reply that used its whole `max_tokens` budget was cut, + /// and says "length"; the hardcoded "stop" read it as finished. + #[tokio::test] + async fn a_reply_cut_at_max_tokens_is_length_not_stop() { + use axum::body::to_bytes; + let resp = build_chat_response( + "cut".to_string(), + None, + false, + 5, + 16, + 16, + std::time::Duration::from_millis(10), + 300.0, + ); + let bytes = to_bytes(resp.into_body(), 64 * 1024).await.expect("body"); + let v: serde_json::Value = serde_json::from_slice(&bytes).expect("json"); + assert_eq!(v["choices"][0]["finish_reason"], "length"); + } + #[tokio::test] async fn build_chat_response_non_streaming_json_body() { use axum::body::to_bytes; @@ -697,6 +728,7 @@ mod chat_helper_tests { false, 5, 3, + 16, std::time::Duration::from_millis(10), 300.0, ); @@ -727,6 +759,7 @@ mod chat_helper_tests { false, 2, 0, + 16, std::time::Duration::from_millis(1), 0.0, ); @@ -748,6 +781,7 @@ mod chat_helper_tests { true, 1, 1, + 16, std::time::Duration::from_millis(1), 1.0, ); @@ -773,6 +807,7 @@ mod chat_helper_tests { true, 1, 1, + 16, std::time::Duration::from_millis(1), 1.0, ); diff --git a/crates/apr-cli/src/commands/serve/context_budget_3718.rs b/crates/apr-cli/src/commands/serve/context_budget_3718.rs new file mode 100644 index 0000000000..bf2b16ad14 --- /dev/null +++ b/crates/apr-cli/src/commands/serve/context_budget_3718.rs @@ -0,0 +1,103 @@ +// #3718 done_when 3: a prompt that does not fit the context window says so, +// never a silent cut. The wgpu handler capped `max_tokens` at 4096 and nothing +// else, so a prompt at or past the context window prefilled anyway and the +// decode loop grew the KV cache past what the model was trained on, reporting +// `finish_reason: "stop"`. These two pure functions are the rule it now follows, +// the same rule the CPU (`effective_max_tokens`) and Qwen3.5 (`Session`) paths +// already enforce. The SafeTensors handlers (chat.rs, simple.rs) apply it too, +// through `st_context_budget`. They are not gated on `wgpu`, so default-feature +// CI tests them. + +/// Tokens the reply may use once the prompt is in a `context_length` window: +/// `min(requested, context_length - prompt_len)`. +/// +/// # Errors +/// `(prompt_len, context_length)` when the prompt leaves no room for even one +/// generated token. That is decided by the request alone, so the caller refuses +/// it as a client error rather than truncating. +#[cfg_attr(not(any(feature = "wgpu", feature = "inference")), allow(dead_code))] +pub(super) fn context_token_budget( + prompt_len: usize, + requested: usize, + context_length: usize, +) -> std::result::Result { + if prompt_len >= context_length { + return Err((prompt_len, context_length)); + } + Ok(requested.min(context_length - prompt_len)) +} + +/// The OpenAI-shaped 400 body for a prompt refused for length +/// (`error.code = "context_length_exceeded"`), so a client can tell it from a +/// server fault without parsing the message. +#[cfg_attr(not(any(feature = "wgpu", feature = "inference")), allow(dead_code))] +pub(super) fn context_length_exceeded_body(prompt_len: usize, context_length: usize) -> serde_json::Value { + serde_json::json!({ + "error": { + "message": format!( + "prompt is {prompt_len} tokens; the model's context window is {context_length}, \ + which leaves no room to generate. The prompt was refused whole, not truncated." + ), + "type": "invalid_request_error", + "param": "messages", + "code": "context_length_exceeded", + "prompt_tokens": prompt_len, + "context_length": context_length, + } + }) +} + +/// OpenAI `finish_reason` for a reply of `generated` tokens under a `max_tokens` +/// budget: `"length"` when the budget ran out, else `"stop"`. Every serve path +/// (wgpu, CUDA, the CUDA->CPU fallback) reports through this one rule, so a cut +/// reply is never labelled a natural stop. +pub(super) fn finish_reason_for(generated: usize, max_tokens: usize) -> &'static str { + if generated >= max_tokens { + "length" + } else { + "stop" + } +} + +#[cfg(test)] +mod tests_context_budget_3718 { + use super::{context_length_exceeded_body, context_token_budget, finish_reason_for}; + + #[test] + fn a_reply_that_used_the_whole_budget_is_length_not_stop() { + assert_eq!(finish_reason_for(64, 64), "length"); + assert_eq!(finish_reason_for(65, 64), "length"); + assert_eq!(finish_reason_for(63, 64), "stop"); + assert_eq!(finish_reason_for(0, 64), "stop"); + } + + #[test] + fn budget_is_the_request_when_it_fits() { + assert_eq!(context_token_budget(10, 64, 2048), Ok(64)); + } + + #[test] + fn budget_is_clamped_to_the_room_left() { + // 2040 prompt tokens in a 2048 window leave 8, whatever was asked. + assert_eq!(context_token_budget(2040, 64, 2048), Ok(8)); + assert_eq!(context_token_budget(2047, 4096, 2048), Ok(1)); + } + + #[test] + fn a_prompt_that_fills_the_window_is_refused_not_cut() { + assert_eq!(context_token_budget(2048, 64, 2048), Err((2048, 2048))); + assert_eq!(context_token_budget(9000, 1, 2048), Err((9000, 2048))); + } + + #[test] + fn refusal_body_names_the_code_and_both_counts() { + let body = context_length_exceeded_body(9000, 2048); + let e = &body["error"]; + assert_eq!(e["code"], "context_length_exceeded"); + assert_eq!(e["type"], "invalid_request_error"); + assert_eq!(e["prompt_tokens"], 9000); + assert_eq!(e["context_length"], 2048); + let msg = e["message"].as_str().expect("message is a string"); + assert!(msg.contains("9000") && msg.contains("2048") && msg.contains("not truncated")); + } +} diff --git a/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs b/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs index 8e83f6b5b9..8d7687e7f4 100644 --- a/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs +++ b/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs @@ -138,7 +138,7 @@ async fn gpu_cpu_fallback( .await; match result { - Ok(Ok(out)) => build_cpu_fallback_response(&out, start), + Ok(Ok(out)) => build_cpu_fallback_response(&out, max_tokens, start), Ok(Err(cpu_err)) => { Json(serde_json::json!({ "error": format!("GPU failed: {gpu_err}; CPU fallback also failed: {cpu_err}") @@ -158,7 +158,11 @@ async fn gpu_cpu_fallback( #[cfg_attr(coverage_nightly, coverage(off))] #[cfg(all(feature = "inference", feature = "cuda"))] #[allow(clippy::disallowed_methods)] -fn build_cpu_fallback_response(out: &AprInferenceOutput, start: Instant) -> axum::response::Response { +fn build_cpu_fallback_response( + out: &AprInferenceOutput, + max_tokens: usize, + start: Instant, +) -> axum::response::Response { use axum::{response::IntoResponse, Json}; let request_id = generate_request_id(); @@ -171,7 +175,8 @@ fn build_cpu_fallback_response(out: &AprInferenceOutput, start: Instant) -> axum "object": "chat.completion", "created": created, "model": "apr-cpu-fallback", - "choices": [{"index": 0, "message": {"role": "assistant", "content": out.text}, "finish_reason": "stop"}], + // #3718: a reply cut at the budget is "length", never "stop". + "choices": [{"index": 0, "message": {"role": "assistant", "content": out.text}, "finish_reason": finish_reason_for(out.tokens_generated, max_tokens)}], "usage": { "prompt_tokens": out.input_token_count, "completion_tokens": out.tokens_generated, @@ -324,7 +329,8 @@ async fn handle_gpu_chat_completion( "object": "chat.completion", "created": created, "model": &response_model, - "choices": [{"index": 0, "message": {"role": "assistant", "content": output_text}, "finish_reason": "stop"}], + // #3718: a reply cut at the budget is "length", never "stop". + "choices": [{"index": 0, "message": {"role": "assistant", "content": output_text}, "finish_reason": finish_reason_for(tokens_generated, max_tokens_clamped)}], "usage": { "prompt_tokens": input_tokens.len(), "completion_tokens": tokens_generated, diff --git a/crates/apr-cli/src/commands/serve/handlers.rs b/crates/apr-cli/src/commands/serve/handlers.rs index afc052b238..fd1b29daff 100644 --- a/crates/apr-cli/src/commands/serve/handlers.rs +++ b/crates/apr-cli/src/commands/serve/handlers.rs @@ -37,6 +37,8 @@ struct WgpuInferenceState { num_layers: usize, vocab_size: usize, hidden_dim: usize, + /// #3718: the model's context window; a prompt that fills it is refused. + context_length: usize, } /// PMAT-355: How one character of a GPT-2 byte-level BPE token maps to bytes. @@ -230,12 +232,13 @@ fn wgpu_stream_done_chunk( id: &str, prompt_len: usize, completion_tokens: u32, + finish_reason: &str, elapsed: std::time::Duration, ) -> String { let tok_s = wgpu_tokens_per_second(completion_tokens as f64, elapsed); serde_json::json!({ "id": id, "object": "chat.completion.chunk", "model": "qwen-wgpu", - "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + "choices": [{"index": 0, "delta": {}, "finish_reason": finish_reason}], "usage": {"prompt_tokens": prompt_len, "completion_tokens": completion_tokens, "total_tokens": prompt_len as u32 + completion_tokens}, "x_wgpu_tok_s": tok_s, @@ -281,7 +284,15 @@ fn wgpu_stream_generate( } } - let done = wgpu_stream_done_chunk(id, prompt_ids.len(), completion_tokens, gen_start.elapsed()); + // #3718: a reply cut at the budget is "length", never "stop". + let finish_reason = finish_reason_for(completion_tokens as usize, max_tokens); + let done = wgpu_stream_done_chunk( + id, + prompt_ids.len(), + completion_tokens, + finish_reason, + gen_start.elapsed(), + ); let _ = tx.blocking_send(done); let _ = tx.blocking_send("[DONE]".to_string()); } @@ -347,11 +358,7 @@ fn wgpu_chat_completion_blocking( .map(|&tok| wgpu_detokenize_one(tok, &state.vocab)) .collect(); let tok_s = wgpu_tokens_per_second(output_ids.len() as f64, elapsed); - let finish_reason = if output_ids.len() >= max_tokens { - "length" - } else { - "stop" - }; + let finish_reason = finish_reason_for(output_ids.len(), max_tokens); axum::Json(serde_json::json!({ "id": id, "object": "chat.completion", "model": "qwen-wgpu", @@ -377,6 +384,20 @@ async fn wgpu_chat_completion( let prompt_ids = wgpu_prompt_ids(&state, &body); let id = wgpu_completion_id(); + // #3718: refuse a prompt that fills the context window, and clamp the budget + // to the room left, before any prefill runs. + let max_tokens = match context_token_budget(prompt_ids.len(), max_tokens, state.context_length) + { + Ok(budget) => budget, + Err((prompt_len, context_length)) => { + return ( + axum::http::StatusCode::BAD_REQUEST, + axum::Json(context_length_exceeded_body(prompt_len, context_length)), + ) + .into_response(); + } + }; + if stream { // PMAT-355: Streaming SSE via spawn_blocking + channel wgpu_chat_completion_streaming(state, prompt_ids, max_tokens, id) @@ -854,6 +875,7 @@ fn serve_wgpu_backend( num_layers, vocab_size, hidden_dim: dims.hidden_dim, + context_length: quantized.config().context_length, }); run_wgpu_server(build_wgpu_router(wgpu_state), config)?; @@ -1715,6 +1737,7 @@ pub fn build_demo_streaming_apr_cpu_router_for_test() -> axum::Router { build_apr_cpu_router(state, super::auth::AuthGate::disabled()) } +include!("context_budget_3718.rs"); include!("handler_apr_cpu_completion.rs"); include!("handler_gpu_completion.rs"); include!("server.rs"); diff --git a/crates/apr-cli/src/commands/serve/safetensors.rs b/crates/apr-cli/src/commands/serve/safetensors.rs index e161493507..97cc4d6f2a 100644 --- a/crates/apr-cli/src/commands/serve/safetensors.rs +++ b/crates/apr-cli/src/commands/serve/safetensors.rs @@ -417,6 +417,35 @@ fn st_cpu_generate( .map_err(|e| e.to_string()) } +/// #3718: the SafeTensors handlers apply the context rule the wgpu handler does, +/// BEFORE generating. Without it `Session` clamped the budget to the room left in +/// the window on its own and the reply was reported `"stop"`, and a prompt that +/// filled the window came back as a 500. Returns the budget to generate with and +/// to judge `finish_reason` against, or the 400 `context_length_exceeded` reply. +#[cfg(feature = "inference")] +fn st_context_budget( + model: &realizar::apr_transformer::AprTransformer, + prompt_len: usize, + max_tokens: usize, +) -> std::result::Result { + use axum::response::IntoResponse; + use realizar::session::ArchForward; + // The same length `Session` checks against (StCpuForward::context_length). + let context_length = realizar::safetensors_infer::StCpuForward::new(model).context_length(); + super::handlers::context_token_budget(prompt_len, max_tokens, context_length).map_err( + |(prompt_len, context_length)| { + ( + axum::http::StatusCode::BAD_REQUEST, + axum::Json(super::handlers::context_length_exceeded_body( + prompt_len, + context_length, + )), + ) + .into_response() + }, + ) +} + #[cfg(all(test, feature = "inference"))] #[path = "tests_st_serve_session_4269.rs"] mod tests_st_serve_session_4269; diff --git a/crates/apr-cli/src/commands/serve/simple.rs b/crates/apr-cli/src/commands/serve/simple.rs index 04ee5b7d41..2ee3060fad 100644 --- a/crates/apr-cli/src/commands/serve/simple.rs +++ b/crates/apr-cli/src/commands/serve/simple.rs @@ -39,7 +39,7 @@ pub(crate) async fn safetensors_generate_handler( .get("temperature") .and_then(|t| t.as_f64()) .unwrap_or(0.0) as f32; - let output_ids = { + let (output_ids, _budget) = { // PMAT-189: Handle transformer lock poisoning gracefully let t = match transformer.lock() { Ok(guard) => guard, @@ -53,8 +53,12 @@ pub(crate) async fn safetensors_generate_handler( .into_response(); } }; - match st_cpu_generate(&t, &input_ids, max_tokens, temperature) { - Ok(ids) => ids, + let budget = match st_context_budget(&t, input_ids.len(), max_tokens) { + Ok(budget) => budget, + Err(refusal) => return refusal, + }; + match st_cpu_generate(&t, &input_ids, budget, temperature) { + Ok(ids) => (ids, budget), Err(e) => { return ( StatusCode::INTERNAL_SERVER_ERROR, diff --git a/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs b/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs index 01a4c627fb..80d6fab715 100644 --- a/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs +++ b/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs @@ -1,7 +1,7 @@ //! #4269: `apr serve` SafeTensors CPU generation goes through //! `realizar::session::Session`, which leaves a witness entry per turn. -use super::st_cpu_generate; +use super::{st_context_budget, st_cpu_generate}; use realizar::apr_transformer::{AprTransformer, AprTransformerConfig, AprTransformerLayer}; use realizar::session::{entries_for, EntryKind}; @@ -51,3 +51,40 @@ fn st_serve_generate_runs_through_session_and_leaves_a_safetensors_witness() { "serve ST CPU generate must go through Session, got {entries:?}" ); } + +/// #3718: a prompt that fills the SafeTensors model's window is refused as a 400 +/// `context_length_exceeded` before `Session` sees it (it used to surface as a 500). +#[tokio::test] +async fn st_prompt_that_fills_the_window_is_a_400_not_a_500() { + let model = tiny_transformer(); // context_length 64 + let refusal = st_context_budget(&model, 64, 8).expect_err("64 tokens fill a 64 window"); + assert_eq!(refusal.status(), axum::http::StatusCode::BAD_REQUEST); + let body = axum::body::to_bytes(refusal.into_body(), usize::MAX) + .await + .expect("body"); + let v: serde_json::Value = serde_json::from_slice(&body).expect("json"); + assert_eq!(v["error"]["code"], "context_length_exceeded"); + assert_eq!(v["error"]["prompt_tokens"], 64); + assert_eq!(v["error"]["context_length"], 64); +} + +/// #3718: near the end of the window the budget is the room left, and that clamped +/// budget is what generation runs with and what `finish_reason` is judged against, +/// so a reply cut by the window reports "length", never "stop". +#[test] +fn st_budget_is_clamped_to_the_room_left_and_generation_stays_inside_it() { + let model = tiny_transformer(); + assert_eq!(st_context_budget(&model, 10, 8).ok(), Some(8)); + let budget = st_context_budget(&model, 60, 64).expect("60 of 64 fits"); + assert_eq!(budget, 4); + let prompt: Vec = (0..60).map(|i| 1 + (i % 11)).collect(); + let out = st_cpu_generate(&model, &prompt, budget, 0.0).expect("generates inside the window"); + let generated = out.len() - prompt.len(); + assert!(generated <= budget, "{generated} > budget {budget}"); + if generated == budget { + assert_eq!( + super::super::handlers::finish_reason_for(generated, budget), + "length" + ); + } +} diff --git a/docs/audits/quorum-PMAT-3718.json b/docs/audits/quorum-PMAT-3718.json index 4c907d2ced..1b6e6301ce 100644 --- a/docs/audits/quorum-PMAT-3718.json +++ b/docs/audits/quorum-PMAT-3718.json @@ -1,16 +1,17 @@ { "ticket": "PMAT-3718", - "base": "225b2a9ab", - "base_resolved": "225b2a9ab", - "base_note": "no origin/225b2a9ab exists; judged against the local ref", - "head": "187b390e1dcfcd6c565d3f27aa779c06a60f8c82", - "diff_sha256": "20f9de301805687f8fb0e6c0853636df2b1aa308c0fbc03c28fe5ff343120b8f", + "base": "origin/main", + "base_resolved": "origin/main", + "base_note": "no origin/origin/main exists; judged against the local ref", + "head": "42eb448d14e6982756d6eca5a5d1a6cb7e1b4628", + "diff_sha256": "6180cac08f8236e7a69e0a5e5610e38f4c28f1449b0cf3c54f6b46f404790a75", "width": 3, "executor": "agy", "prompt_mode": "inline", - "prompt_bytes": 65794, + "prompt_bytes": 49675, + "prompt_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", "author": { - "model": "claude-opus-5", + "model": "claude-opus-5-5", "family": "claude", "source": "flag" }, @@ -20,50 +21,41 @@ "lane": 1, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff successfully implements PMAT-3718, correctly adding `prompt_tokens`, `completion_tokens`, and `finish_reason` to `apr run --json` and fixing `apr serve` endpoints to accurately report truncations. It resolves the silent stream cuts for clamped requests by judging the finish against the actual ran budget, maps unfittable prompts to a proper 400 error, and stops the APR CPU handler from echoing the prompt when no tokens are generated. The implementation conforms to all acceptance criteria and the provided receipt claims.", + "summary": "The diff correctly implements the \"serve\" half of PMAT-3718: a new finish_reason_for(generated, max_tokens) rule (\"length\" when generated >= the budget actually run with, else \"stop\") replaces every hardcoded \"stop\" in the SafeTensors chat handler, the wgpu non-streaming/streaming handlers, and the CUDA/CPU-fallback handlers. A new context_token_budget helper clamps max_tokens to the room left in the context window and refuses (400 context_length_exceeded) a prompt that fills it, wired into both the wgpu and SafeTensors paths before generation runs, matching done_when 3. Verified by reading (not running) the code: build_chat_response's finish_reason is judged against the clamped budget actually passed to st_cpu_generate (not the raw request), StCpuForward::context_length() exists and matches the comment's claim, and the tiny_transformer fixture used in the new 400-refusal test does have context_length 64 as asserted. No test asserts the opposite of the ticket, no gate is weakened, and I found no scope creep beyond what the ticket describes.", "findings": [ { - "claim": "Adds accurate token counts and finish_reason to CLI usage output using the actual engine count.", - "file": "crates/apr-cli/src/commands/inference_output.rs", - "grounding": "cited", - "line": 184 - }, - { - "claim": "Shadows max_tokens with clamped budget for correct length evaluation, and maps unfittable prompt errors to a 400 response.", - "file": "crates/aprender-serve/src/api/cuda_chat_backend.rs", - "grounding": "cited", - "line": 322 - }, - { - "claim": "Implements budget-aware finish reason determination correctly handling clamped runs and intrinsic stop token consumption.", - "file": "crates/aprender-serve/src/infer/run_report.rs", - "grounding": "cited", - "line": 39 - }, - { - "claim": "Fixes APR CPU fallback to yield empty instead of the prompt when nothing is generated, and uses `from_decode` to determine stop reason accurately.", - "file": "crates/apr-cli/src/commands/serve/handlers.rs", - "grounding": "cited", - "line": 1129 + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, + "claim": "Pre-existing, not introduced by this diff: build_gpu_sse_stream's terminal [DONE] SSE event carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop.", + "fix": "Out of scope for this diff (untouched file region) but worth a follow-up ticket if the CUDA streaming path is meant to satisfy the same #3718 guarantee as the wgpu streaming path, which does send a terminal chunk with finish_reason.", + "grounding": "cited" } ], - "raw_bytes": 4150, + "raw_bytes": 4281, "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-high", - "model_measured": "gemini-3.1-pro-high", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "cddc014e8538193133956da3e3252cb1fddeefe98d8a08941e3d14ac1d1f2859", + "output_sha256": "331b524226a250425a582f87c748927dca1bb4a7d0a7780fdc1aa0484df231d0", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-high", + "judged_by": "claude-sonnet-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-high", - "family": "gemini", + "model": "claude-sonnet-5", + "family": "claude", "outcome": "answered" } ], @@ -74,54 +66,76 @@ "lane": 2, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff strictly conforms to the acceptance criteria of PMAT-3718 and makes no unauthorized changes. `RunUsage` is implemented and emitted by `apr run --json` with `null` fallbacks. `FinishReason` correctly distinguishes cuts from stops via `from_decode`, fixing silent cuts in both CPU non-streaming and streaming `serve` paths. The context-clamped budgets are correctly used to evaluate the finish reason for the dense CPU loop, and over-long prompts now return a 400 error. The `a_fixed_text_prompt_counts_the_tokens_the_tokenizer_feeds` test adequately verifies the token count via the newly exposed `build_executable_pygmy_gguf_with` metadata hook. All receipt claims are backed by the code and tests.", + "summary": "Diff correctly implements PMAT-3718's serve-side requirement: a new finish_reason_for(generated, max_tokens) rule replaces hardcoded \"stop\" in SafeTensors chat/simple handlers, the CUDA non-streaming path, the CUDA→CPU fallback, and both wgpu streaming/non-streaming paths. A new context_token_budget clamps generation and refuses over-long prompts with 400 context_length_exceeded (SafeTensors via st_context_budget, wgpu via context_token_budget), applied before generation runs so the budget used for generation is the same one finish_reason is judged against. Traced the generated/max_tokens pairing through chat.rs, safetensors.rs, handler_gpu_completion.rs and handlers.rs — all internally consistent, no mismatched variables. New/updated tests match the intended behavior; no test asserts the opposite of the ticket. The \"apr run --json\" half of the ticket is already present on this branch (run.rs, run_tests_usage_3718.rs) from a prior commit, consistent with the 50%-InProgress status. handler_apr_cpu_completion.rs was left untouched but already applies the same length/stop rule inline (pre-existing), so it's not a gap. No gate weakening, no scope creep found.", "findings": [], - "raw_bytes": 2788, - "err_bytes": 100, + "raw_bytes": 3288, + "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-low", - "model_measured": "gemini-3.1-pro-low", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "e1f123825b8ff98955b709a499a6fe1b4eb2a8ba60ad9c2de24eae0772d2753b", + "output_sha256": "fa75e96bb14b90e3f0693b5f3029a2568d7339241a6ad15a4917b11df8a4157b", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-low", + "judged_by": "claude-sonnet-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-low", - "family": "gemini", + "model": "claude-sonnet-5", + "family": "claude", "outcome": "answered" } ], - "skipped": [] - } + "skipped": [ + { + "model": "gemini-3.1-pro-high", + "why": "family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ] + }, + "repeat_of": 1 }, { "lane": 3, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff perfectly implements the requested features from PMAT-3718. It adds `prompt_tokens`, `completion_tokens`, `finish_reason`, and `context_length` to `apr run --json` using the new `RunUsage` structure and the `run_inference_report` hook, returning `null` when a specific backend does not report these metrics. \nFor `apr serve`, it correctly calculates the `finish_reason` in `/v1/chat/completions` against the effective clamped budget rather than the requested one, preventing context-clamped responses from being falsely reported as `stop`. It correctly throws a 400 error (instead of truncating or throwing a 500 after the fact) when the prompt exceeds the context window. \nThe comprehensive test additions and updates enforce these behaviors properly, including the exact `a_fixed_text_prompt_counts_the_tokens_the_tokenizer_feeds` case row required by the ticket's done criteria.", + "summary": "Diff correctly implements PMAT-3718's serve half: replaces all hardcoded \"stop\" finish reasons with finish_reason_for(generated, max_tokens)→\"length\" when generated≥max_tokens, else \"stop\". Updates SafeTensors (chat/simple), wgpu (streaming/non-streaming), and CUDA (non-streaming/CPU-fallback) paths. Adds context budget validation that returns 400 context_length_exceeded before prefill when prompt fills window. Tests verify boundary cases and SafeTensors integration. No gates weakened, no unauthorized changes.", "findings": [], - "raw_bytes": 3040, + "raw_bytes": 1969, "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-high", - "model_measured": "gemini-3.1-pro-high", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-haiku-4-5", + "model_measured": "claude-haiku-4-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "e090c837b5c14541d2a7899cc36a5f7bca39b8aef258c86b0ec3474fb65032a3", + "output_sha256": "7d5caa1c1ea251c78d226a68d9cee94e087b4d4f5ffa9535406729e8d2f29ff2", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-high", + "judged_by": "claude-haiku-4-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-high", - "family": "gemini", + "model": "claude-haiku-4-5", + "family": "claude", "outcome": "answered" } ], @@ -132,43 +146,13 @@ "dissent": [], "dedup": [ { - "file": "crates/apr-cli/src/commands/inference_output.rs", - "line": 184, + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, "lanes_agreeing": [ 1 ], "claims": [ - "Adds accurate token counts and finish_reason to CLI usage output using the actual engine count." - ] - }, - { - "file": "crates/apr-cli/src/commands/serve/handlers.rs", - "line": 1129, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Fixes APR CPU fallback to yield empty instead of the prompt when nothing is generated, and uses `from_decode` to determine stop reason accurately." - ] - }, - { - "file": "crates/aprender-serve/src/api/cuda_chat_backend.rs", - "line": 322, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Shadows max_tokens with clamped budget for correct length evaluation, and maps unfittable prompt errors to a 400 response." - ] - }, - { - "file": "crates/aprender-serve/src/infer/run_report.rs", - "line": 39, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Implements budget-aware finish reason determination correctly handling clamped runs and intrinsic stop token consumption." + "Pre-existing, not introduced by this diff: build_gpu_sse_stream's terminal [DONE] SSE event carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop." ] } ], @@ -176,66 +160,178 @@ "coverage_source": "lanes", "partial": false, "partial_reasons": [ - "lane models: only 2 distinct model ids across 3 lanes (gemini-3.1-pro-high, gemini-3.1-pro-low) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)" + "lane models: only 2 distinct model ids across 3 lanes (claude-haiku-4-5, claude-sonnet-5) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)", + "lane 2: judged by claude-sonnet-5 after falling through gemini-3.1-pro-high (skipped: family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24) (PMAT-321)" ], "fallback": { - "same_family_width": 1, + "same_family_width": 2, "chain": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, { "model": "gemini-3.1-pro-high", "family": "gemini", "disposition": "configured" }, { - "model": "gemini-3.1-pro-low", - "family": "gemini", - "disposition": "configured" + "model": "claude-haiku-4-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" }, { - "model": "gpt-oss-120b-medium", - "family": "openai", - "disposition": "fallback" + "model": "claude-opus-5-5", + "family": "claude", + "disposition": "excluded-self-review", + "why": "the author's own model id (claude-opus-5-5) — a model never reviews its own diff, at any width (R-15a identity bar)" }, { "model": "qwen3.5", "family": "qwen", "disposition": "not-run", "why": "no quorum.local_lane in the config — the aprender lane has no model to load" - }, - { - "model": "claude-opus-4-6-thinking", - "family": "claude", - "disposition": "width", - "why": "same family as the author: at most 1 lane, recorded role width, counted toward no floor (R-15a)" - }, - { - "model": "claude-sonnet-4-6", - "family": "claude", - "disposition": "width", - "why": "same family as the author: at most 1 lane, recorded role width, counted toward no floor (R-15a)" } ], "precheck": [ { "family": "gemini", "model": "gemini-3.1-pro-high", - "probe": 1, - "outcome": "live" + "probe": 0, + "outcome": "policy", + "reason": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" } ], + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, "prah": { "source": "install-receipt", "path": "/home/noah/.claude/skills/paiml-implement/bin/prah" + }, + "tier1": { + "agy_tier1_only": true, + "tier1": false, + "matched": [], + "paths": 8, + "patterns": [ + "^\\.github/workflows/", + "(^|/)release[^/]*\\.(ya?ml|sh|rs|toml)$", + "cuda|kernel|\\.cu$|\\.ptx$", + "(^|/)[^/]*(gate|guard)[^/]*\\.(sh|rs|py)$|^hooks/", + "security|secret|credential|(^|/)deny\\.toml$", + "^skills/quorum-review/|(^|/)(receipt-lint|roadmap-lint|release-lint|kind-gate|model-gate)[^/]*$|^crates/prah-lint/" + ], + "builtin": "^skills/quorum-review/|(^|/)skills/paiml-implement/config\\.json$|(^|/)(modellib|lane-reduce|lane-fallback|lane-group|agy-lane|cc-lane|receipt-lint|route|quota)\\.sh$|^crates/prah-lint/" + }, + "bucket": { + "ledger": "/home/noah/.local/state/paiml-implement/agy-bucket.jsonl", + "window_s": 18000, + "pace": "off", + "buckets": {}, + "closed": [] } }, "auto_merge": { - "checked": false, + "checked": true, "was_armed": false, "disarmed": false, - "note": "no --pr given: nothing to disarm" + "note": "auto-merge not armed" + }, + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "cheap_seat": "claude-haiku-4-5", + "canary": { + "skipped": "brief 49675 bytes > 24576" + }, + "advisory_lane": { + "state": "answered", + "counts": false, + "row": { + "Verdict": { + "verdict": "PASS", + "cell": "gx10-cuda", + "backend": "cuda" + } + }, + "verdict": "PASS", + "why": null, + "served_by": "gx10-cuda", + "gpu_proof": { + "used_gpu_probe": true, + "server_holds_gpu": true, + "used_gpu_source": "completions probe (chat omits used_gpu, aprender#4146)", + "trace_lines": [ + "3340528, 309 MiB" + ] + }, + "apr": "0.70.0", + "apr_binary_sha256": "6b2a7dc66c38adcca20b315beac70a1a31e7bd0af2a1cc5c2a7730ca6086d417", + "apr_tag": "v0.70.0-dev.3990584f0", + "apr_commit": "3990584f0bf5af6bca34ea9298df92177f3ef79a", + "apr_build_why": null, + "model": "/home/noah/data/models/Qwen3.5-4B-Q4_K_M.gguf", + "model_sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "rc": 0, + "wall_s": 15, + "collected_s": 821, + "budget_s": 128, + "budget_basis": "ledger: 2 x p95 64 s over 27 gx10-cuda Verdict rows", + "brief": { + "bytes": 49675, + "sent_bytes": 49844, + "max_bytes": 196608 + }, + "trace": { + "input_sha256": "568b8499c5f3b47f6bfce98326c10139208625a14c5b0a13ef052244dd0415c0", + "raw": "advisory.json" + }, + "attempts": [ + { + "host": "gx10", + "ok": true, + "wall_s": 15, + "used_gpu": true, + "ts": "2026-09-27T21:44:23.578Z", + "wall_ms": 15010, + "request_id": "01a0e4d3-8917-716f-b923-666910d9a5ea", + "load1": 12.20, + "cold": false + } + ], + "ledger": "/home/noah/.local/state/paiml-implement/advisory-ledger.jsonl", + "raw": "advisory.json", + "agrees_with_counted": true, + "counted": "PASS" }, "lint": { "ok": true, - "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5/claude" + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5-5/claude same_family=3/2 degraded=same-family (policy: agy reserved for tier-1 diffs)" } }