diff --git a/crates/apr-cli/src/commands/serve/chat.rs b/crates/apr-cli/src/commands/serve/chat.rs index ef240b175d..b0ff459b02 100644 --- a/crates/apr-cli/src/commands/serve/chat.rs +++ b/crates/apr-cli/src/commands/serve/chat.rs @@ -266,7 +266,7 @@ pub(crate) async fn safetensors_chat_completions_handler( .get("temperature") .and_then(|t| t.as_f64()) .unwrap_or(0.0) as f32; - let output_ids = { + let (output_ids, max_tokens) = { // PMAT-189: Handle transformer lock poisoning gracefully let t = match transformer.lock() { Ok(guard) => guard, @@ -280,8 +280,12 @@ pub(crate) async fn safetensors_chat_completions_handler( .into_response(); } }; - match st_cpu_generate(&t, &input_ids, max_tokens, temperature) { - Ok(ids) => ids, + let budget = match st_context_budget(&t, input_ids.len(), max_tokens) { + Ok(budget) => budget, + Err(refusal) => return refusal, + }; + match st_cpu_generate(&t, &input_ids, budget, temperature) { + Ok(ids) => (ids, budget), Err(e) => { return ( StatusCode::INTERNAL_SERVER_ERROR, @@ -336,6 +340,7 @@ pub(crate) async fn safetensors_chat_completions_handler( stream_mode, input_ids.len(), tokens_generated, + max_tokens, elapsed, tok_per_sec, ) @@ -419,6 +424,7 @@ fn build_chat_response( stream_mode: bool, prompt_tokens: usize, tokens_generated: usize, + max_tokens: usize, elapsed: std::time::Duration, tok_per_sec: f64, ) -> axum::response::Response { @@ -427,7 +433,12 @@ fn build_chat_response( let request_id = generate_request_id(); let has_tool_calls = tool_calls.is_some(); - let finish_reason = if has_tool_calls { "tool_calls" } else { "stop" }; + // #3718: a reply cut at `max_tokens` is "length", never "stop". + let finish_reason = if has_tool_calls { + "tool_calls" + } else { + super::handlers::finish_reason_for(tokens_generated, max_tokens) + }; let created = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .unwrap_or_default() @@ -688,6 +699,26 @@ mod chat_helper_tests { // ---- build_chat_response ------------------------------------------ + /// #3718: a SafeTensors reply that used its whole `max_tokens` budget was cut, + /// and says "length"; the hardcoded "stop" read it as finished. + #[tokio::test] + async fn a_reply_cut_at_max_tokens_is_length_not_stop() { + use axum::body::to_bytes; + let resp = build_chat_response( + "cut".to_string(), + None, + false, + 5, + 16, + 16, + std::time::Duration::from_millis(10), + 300.0, + ); + let bytes = to_bytes(resp.into_body(), 64 * 1024).await.expect("body"); + let v: serde_json::Value = serde_json::from_slice(&bytes).expect("json"); + assert_eq!(v["choices"][0]["finish_reason"], "length"); + } + #[tokio::test] async fn build_chat_response_non_streaming_json_body() { use axum::body::to_bytes; @@ -697,6 +728,7 @@ mod chat_helper_tests { false, 5, 3, + 16, std::time::Duration::from_millis(10), 300.0, ); @@ -727,6 +759,7 @@ mod chat_helper_tests { false, 2, 0, + 16, std::time::Duration::from_millis(1), 0.0, ); @@ -748,6 +781,7 @@ mod chat_helper_tests { true, 1, 1, + 16, std::time::Duration::from_millis(1), 1.0, ); @@ -773,6 +807,7 @@ mod chat_helper_tests { true, 1, 1, + 16, std::time::Duration::from_millis(1), 1.0, ); diff --git a/crates/apr-cli/src/commands/serve/context_budget_3718.rs b/crates/apr-cli/src/commands/serve/context_budget_3718.rs new file mode 100644 index 0000000000..bf2b16ad14 --- /dev/null +++ b/crates/apr-cli/src/commands/serve/context_budget_3718.rs @@ -0,0 +1,103 @@ +// #3718 done_when 3: a prompt that does not fit the context window says so, +// never a silent cut. The wgpu handler capped `max_tokens` at 4096 and nothing +// else, so a prompt at or past the context window prefilled anyway and the +// decode loop grew the KV cache past what the model was trained on, reporting +// `finish_reason: "stop"`. These two pure functions are the rule it now follows, +// the same rule the CPU (`effective_max_tokens`) and Qwen3.5 (`Session`) paths +// already enforce. The SafeTensors handlers (chat.rs, simple.rs) apply it too, +// through `st_context_budget`. They are not gated on `wgpu`, so default-feature +// CI tests them. + +/// Tokens the reply may use once the prompt is in a `context_length` window: +/// `min(requested, context_length - prompt_len)`. +/// +/// # Errors +/// `(prompt_len, context_length)` when the prompt leaves no room for even one +/// generated token. That is decided by the request alone, so the caller refuses +/// it as a client error rather than truncating. +#[cfg_attr(not(any(feature = "wgpu", feature = "inference")), allow(dead_code))] +pub(super) fn context_token_budget( + prompt_len: usize, + requested: usize, + context_length: usize, +) -> std::result::Result { + if prompt_len >= context_length { + return Err((prompt_len, context_length)); + } + Ok(requested.min(context_length - prompt_len)) +} + +/// The OpenAI-shaped 400 body for a prompt refused for length +/// (`error.code = "context_length_exceeded"`), so a client can tell it from a +/// server fault without parsing the message. +#[cfg_attr(not(any(feature = "wgpu", feature = "inference")), allow(dead_code))] +pub(super) fn context_length_exceeded_body(prompt_len: usize, context_length: usize) -> serde_json::Value { + serde_json::json!({ + "error": { + "message": format!( + "prompt is {prompt_len} tokens; the model's context window is {context_length}, \ + which leaves no room to generate. The prompt was refused whole, not truncated." + ), + "type": "invalid_request_error", + "param": "messages", + "code": "context_length_exceeded", + "prompt_tokens": prompt_len, + "context_length": context_length, + } + }) +} + +/// OpenAI `finish_reason` for a reply of `generated` tokens under a `max_tokens` +/// budget: `"length"` when the budget ran out, else `"stop"`. Every serve path +/// (wgpu, CUDA, the CUDA->CPU fallback) reports through this one rule, so a cut +/// reply is never labelled a natural stop. +pub(super) fn finish_reason_for(generated: usize, max_tokens: usize) -> &'static str { + if generated >= max_tokens { + "length" + } else { + "stop" + } +} + +#[cfg(test)] +mod tests_context_budget_3718 { + use super::{context_length_exceeded_body, context_token_budget, finish_reason_for}; + + #[test] + fn a_reply_that_used_the_whole_budget_is_length_not_stop() { + assert_eq!(finish_reason_for(64, 64), "length"); + assert_eq!(finish_reason_for(65, 64), "length"); + assert_eq!(finish_reason_for(63, 64), "stop"); + assert_eq!(finish_reason_for(0, 64), "stop"); + } + + #[test] + fn budget_is_the_request_when_it_fits() { + assert_eq!(context_token_budget(10, 64, 2048), Ok(64)); + } + + #[test] + fn budget_is_clamped_to_the_room_left() { + // 2040 prompt tokens in a 2048 window leave 8, whatever was asked. + assert_eq!(context_token_budget(2040, 64, 2048), Ok(8)); + assert_eq!(context_token_budget(2047, 4096, 2048), Ok(1)); + } + + #[test] + fn a_prompt_that_fills_the_window_is_refused_not_cut() { + assert_eq!(context_token_budget(2048, 64, 2048), Err((2048, 2048))); + assert_eq!(context_token_budget(9000, 1, 2048), Err((9000, 2048))); + } + + #[test] + fn refusal_body_names_the_code_and_both_counts() { + let body = context_length_exceeded_body(9000, 2048); + let e = &body["error"]; + assert_eq!(e["code"], "context_length_exceeded"); + assert_eq!(e["type"], "invalid_request_error"); + assert_eq!(e["prompt_tokens"], 9000); + assert_eq!(e["context_length"], 2048); + let msg = e["message"].as_str().expect("message is a string"); + assert!(msg.contains("9000") && msg.contains("2048") && msg.contains("not truncated")); + } +} diff --git a/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs b/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs index 8e83f6b5b9..8d7687e7f4 100644 --- a/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs +++ b/crates/apr-cli/src/commands/serve/handler_gpu_completion.rs @@ -138,7 +138,7 @@ async fn gpu_cpu_fallback( .await; match result { - Ok(Ok(out)) => build_cpu_fallback_response(&out, start), + Ok(Ok(out)) => build_cpu_fallback_response(&out, max_tokens, start), Ok(Err(cpu_err)) => { Json(serde_json::json!({ "error": format!("GPU failed: {gpu_err}; CPU fallback also failed: {cpu_err}") @@ -158,7 +158,11 @@ async fn gpu_cpu_fallback( #[cfg_attr(coverage_nightly, coverage(off))] #[cfg(all(feature = "inference", feature = "cuda"))] #[allow(clippy::disallowed_methods)] -fn build_cpu_fallback_response(out: &AprInferenceOutput, start: Instant) -> axum::response::Response { +fn build_cpu_fallback_response( + out: &AprInferenceOutput, + max_tokens: usize, + start: Instant, +) -> axum::response::Response { use axum::{response::IntoResponse, Json}; let request_id = generate_request_id(); @@ -171,7 +175,8 @@ fn build_cpu_fallback_response(out: &AprInferenceOutput, start: Instant) -> axum "object": "chat.completion", "created": created, "model": "apr-cpu-fallback", - "choices": [{"index": 0, "message": {"role": "assistant", "content": out.text}, "finish_reason": "stop"}], + // #3718: a reply cut at the budget is "length", never "stop". + "choices": [{"index": 0, "message": {"role": "assistant", "content": out.text}, "finish_reason": finish_reason_for(out.tokens_generated, max_tokens)}], "usage": { "prompt_tokens": out.input_token_count, "completion_tokens": out.tokens_generated, @@ -324,7 +329,8 @@ async fn handle_gpu_chat_completion( "object": "chat.completion", "created": created, "model": &response_model, - "choices": [{"index": 0, "message": {"role": "assistant", "content": output_text}, "finish_reason": "stop"}], + // #3718: a reply cut at the budget is "length", never "stop". + "choices": [{"index": 0, "message": {"role": "assistant", "content": output_text}, "finish_reason": finish_reason_for(tokens_generated, max_tokens_clamped)}], "usage": { "prompt_tokens": input_tokens.len(), "completion_tokens": tokens_generated, diff --git a/crates/apr-cli/src/commands/serve/handlers.rs b/crates/apr-cli/src/commands/serve/handlers.rs index afc052b238..fd1b29daff 100644 --- a/crates/apr-cli/src/commands/serve/handlers.rs +++ b/crates/apr-cli/src/commands/serve/handlers.rs @@ -37,6 +37,8 @@ struct WgpuInferenceState { num_layers: usize, vocab_size: usize, hidden_dim: usize, + /// #3718: the model's context window; a prompt that fills it is refused. + context_length: usize, } /// PMAT-355: How one character of a GPT-2 byte-level BPE token maps to bytes. @@ -230,12 +232,13 @@ fn wgpu_stream_done_chunk( id: &str, prompt_len: usize, completion_tokens: u32, + finish_reason: &str, elapsed: std::time::Duration, ) -> String { let tok_s = wgpu_tokens_per_second(completion_tokens as f64, elapsed); serde_json::json!({ "id": id, "object": "chat.completion.chunk", "model": "qwen-wgpu", - "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + "choices": [{"index": 0, "delta": {}, "finish_reason": finish_reason}], "usage": {"prompt_tokens": prompt_len, "completion_tokens": completion_tokens, "total_tokens": prompt_len as u32 + completion_tokens}, "x_wgpu_tok_s": tok_s, @@ -281,7 +284,15 @@ fn wgpu_stream_generate( } } - let done = wgpu_stream_done_chunk(id, prompt_ids.len(), completion_tokens, gen_start.elapsed()); + // #3718: a reply cut at the budget is "length", never "stop". + let finish_reason = finish_reason_for(completion_tokens as usize, max_tokens); + let done = wgpu_stream_done_chunk( + id, + prompt_ids.len(), + completion_tokens, + finish_reason, + gen_start.elapsed(), + ); let _ = tx.blocking_send(done); let _ = tx.blocking_send("[DONE]".to_string()); } @@ -347,11 +358,7 @@ fn wgpu_chat_completion_blocking( .map(|&tok| wgpu_detokenize_one(tok, &state.vocab)) .collect(); let tok_s = wgpu_tokens_per_second(output_ids.len() as f64, elapsed); - let finish_reason = if output_ids.len() >= max_tokens { - "length" - } else { - "stop" - }; + let finish_reason = finish_reason_for(output_ids.len(), max_tokens); axum::Json(serde_json::json!({ "id": id, "object": "chat.completion", "model": "qwen-wgpu", @@ -377,6 +384,20 @@ async fn wgpu_chat_completion( let prompt_ids = wgpu_prompt_ids(&state, &body); let id = wgpu_completion_id(); + // #3718: refuse a prompt that fills the context window, and clamp the budget + // to the room left, before any prefill runs. + let max_tokens = match context_token_budget(prompt_ids.len(), max_tokens, state.context_length) + { + Ok(budget) => budget, + Err((prompt_len, context_length)) => { + return ( + axum::http::StatusCode::BAD_REQUEST, + axum::Json(context_length_exceeded_body(prompt_len, context_length)), + ) + .into_response(); + } + }; + if stream { // PMAT-355: Streaming SSE via spawn_blocking + channel wgpu_chat_completion_streaming(state, prompt_ids, max_tokens, id) @@ -854,6 +875,7 @@ fn serve_wgpu_backend( num_layers, vocab_size, hidden_dim: dims.hidden_dim, + context_length: quantized.config().context_length, }); run_wgpu_server(build_wgpu_router(wgpu_state), config)?; @@ -1715,6 +1737,7 @@ pub fn build_demo_streaming_apr_cpu_router_for_test() -> axum::Router { build_apr_cpu_router(state, super::auth::AuthGate::disabled()) } +include!("context_budget_3718.rs"); include!("handler_apr_cpu_completion.rs"); include!("handler_gpu_completion.rs"); include!("server.rs"); diff --git a/crates/apr-cli/src/commands/serve/safetensors.rs b/crates/apr-cli/src/commands/serve/safetensors.rs index e161493507..d538a3d281 100644 --- a/crates/apr-cli/src/commands/serve/safetensors.rs +++ b/crates/apr-cli/src/commands/serve/safetensors.rs @@ -417,10 +417,43 @@ fn st_cpu_generate( .map_err(|e| e.to_string()) } +/// #3718: the SafeTensors handlers apply the context rule the wgpu handler does, +/// BEFORE generating. Without it `Session` clamped the budget to the room left in +/// the window on its own and the reply was reported `"stop"`, and a prompt that +/// filled the window came back as a 500. Returns the budget to generate with and +/// to judge `finish_reason` against, or the 400 `context_length_exceeded` reply. +#[cfg(feature = "inference")] +fn st_context_budget( + model: &realizar::apr_transformer::AprTransformer, + prompt_len: usize, + max_tokens: usize, +) -> std::result::Result { + use axum::response::IntoResponse; + use realizar::session::ArchForward; + // The same length `Session` checks against (StCpuForward::context_length). + let context_length = realizar::safetensors_infer::StCpuForward::new(model).context_length(); + super::handlers::context_token_budget(prompt_len, max_tokens, context_length).map_err( + |(prompt_len, context_length)| { + ( + axum::http::StatusCode::BAD_REQUEST, + axum::Json(super::handlers::context_length_exceeded_body( + prompt_len, + context_length, + )), + ) + .into_response() + }, + ) +} + #[cfg(all(test, feature = "inference"))] #[path = "tests_st_serve_session_4269.rs"] mod tests_st_serve_session_4269; +#[cfg(all(test, feature = "inference"))] +#[path = "tests_st_overlength_router_3718.rs"] +mod tests_st_overlength_router_3718; + /// #3979: the SafeTensors HTTP surface, in ONE place. It was assembled inline twice /// (single-file and sharded), differing only in the `/tensors` payload. Every route is /// mounted AND recorded, so `GET /` and the 404 list exactly what is served; the diff --git a/crates/apr-cli/src/commands/serve/simple.rs b/crates/apr-cli/src/commands/serve/simple.rs index 04ee5b7d41..2ee3060fad 100644 --- a/crates/apr-cli/src/commands/serve/simple.rs +++ b/crates/apr-cli/src/commands/serve/simple.rs @@ -39,7 +39,7 @@ pub(crate) async fn safetensors_generate_handler( .get("temperature") .and_then(|t| t.as_f64()) .unwrap_or(0.0) as f32; - let output_ids = { + let (output_ids, _budget) = { // PMAT-189: Handle transformer lock poisoning gracefully let t = match transformer.lock() { Ok(guard) => guard, @@ -53,8 +53,12 @@ pub(crate) async fn safetensors_generate_handler( .into_response(); } }; - match st_cpu_generate(&t, &input_ids, max_tokens, temperature) { - Ok(ids) => ids, + let budget = match st_context_budget(&t, input_ids.len(), max_tokens) { + Ok(budget) => budget, + Err(refusal) => return refusal, + }; + match st_cpu_generate(&t, &input_ids, budget, temperature) { + Ok(ids) => (ids, budget), Err(e) => { return ( StatusCode::INTERNAL_SERVER_ERROR, diff --git a/crates/apr-cli/src/commands/serve/tests_st_overlength_router_3718.rs b/crates/apr-cli/src/commands/serve/tests_st_overlength_router_3718.rs new file mode 100644 index 0000000000..7113826f48 --- /dev/null +++ b/crates/apr-cli/src/commands/serve/tests_st_overlength_router_3718.rs @@ -0,0 +1,62 @@ +//! #3718 vs PMAT-4616 (D5, #4614): the over-length refusal in `apr serve`'s OWN +//! SafeTensors handlers. D5 maps `aprender-serve`'s api errors to 4xx; it does not +//! touch these apr-cli handlers, so on `main` (with or without D5) this request is a +//! 500 "Generation failed". Driven through the REAL `safetensors_app` router. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] + +use super::tests_st_serve_session_4269::tiny_transformer; +use super::{safetensors_app, SafeTensorsState}; +use axum::body::Body; +use axum::http::{Method, Request, StatusCode}; +use std::sync::{Arc, Mutex}; +use tower::ServiceExt; + +fn router() -> axum::Router { + let state = SafeTensorsState { + transformer: Some(Arc::new(Mutex::new(tiny_transformer()))), // context_length 64 + tokenizer_info: None, // prompt chars become token ids, one per char + model_path: "tiny.safetensors".into(), + }; + safetensors_app(serde_json::json!({"count": 0}), true, state, "tiny".into()) +} + +async fn post(path: &str, body: serde_json::Value) -> (StatusCode, serde_json::Value) { + let req = Request::builder() + .method(Method::POST) + .uri(path) + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap(); + let resp = router().oneshot(req).await.unwrap(); + let status = resp.status(); + let bytes = axum::body::to_bytes(resp.into_body(), usize::MAX) + .await + .unwrap(); + let v = serde_json::from_slice(&bytes) + .unwrap_or_else(|_| panic!("{status}: {}", String::from_utf8_lossy(&bytes))); + (status, v) +} + +/// 200 prompt tokens cannot fit a 64-token window: a client error, named as such. +fn assert_context_refusal(path: &str, status: StatusCode, v: &serde_json::Value) { + assert_eq!(status, StatusCode::BAD_REQUEST, "{path}: {v}"); + assert_eq!(v["error"]["code"], "context_length_exceeded", "{path}: {v}"); +} + +#[tokio::test] +async fn st_router_generate_over_length_is_a_400_context_length_exceeded() { + let body = serde_json::json!({"prompt": "a".repeat(200), "max_tokens": 4}); + let (st, v) = post("/generate", body).await; + assert_context_refusal("/generate", st, &v); +} + +#[tokio::test] +async fn st_router_chat_over_length_is_a_400_context_length_exceeded() { + let body = serde_json::json!({ + "model": "tiny", + "messages": [{"role": "user", "content": "a".repeat(200)}], + "max_tokens": 4, + }); + let (st, v) = post("/v1/chat/completions", body).await; + assert_context_refusal("/v1/chat/completions", st, &v); +} diff --git a/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs b/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs index 01a4c627fb..653000a665 100644 --- a/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs +++ b/crates/apr-cli/src/commands/serve/tests_st_serve_session_4269.rs @@ -1,11 +1,11 @@ //! #4269: `apr serve` SafeTensors CPU generation goes through //! `realizar::session::Session`, which leaves a witness entry per turn. -use super::st_cpu_generate; +use super::{st_context_budget, st_cpu_generate}; use realizar::apr_transformer::{AprTransformer, AprTransformerConfig, AprTransformerLayer}; use realizar::session::{entries_for, EntryKind}; -fn tiny_transformer() -> AprTransformer { +pub(super) fn tiny_transformer() -> AprTransformer { let (hidden_dim, intermediate_dim, vocab_size) = (8, 16, 12); let config = AprTransformerConfig { architecture: "safetensors-test".to_string(), @@ -51,3 +51,40 @@ fn st_serve_generate_runs_through_session_and_leaves_a_safetensors_witness() { "serve ST CPU generate must go through Session, got {entries:?}" ); } + +/// #3718: a prompt that fills the SafeTensors model's window is refused as a 400 +/// `context_length_exceeded` before `Session` sees it (it used to surface as a 500). +#[tokio::test] +async fn st_prompt_that_fills_the_window_is_a_400_not_a_500() { + let model = tiny_transformer(); // context_length 64 + let refusal = st_context_budget(&model, 64, 8).expect_err("64 tokens fill a 64 window"); + assert_eq!(refusal.status(), axum::http::StatusCode::BAD_REQUEST); + let body = axum::body::to_bytes(refusal.into_body(), usize::MAX) + .await + .expect("body"); + let v: serde_json::Value = serde_json::from_slice(&body).expect("json"); + assert_eq!(v["error"]["code"], "context_length_exceeded"); + assert_eq!(v["error"]["prompt_tokens"], 64); + assert_eq!(v["error"]["context_length"], 64); +} + +/// #3718: near the end of the window the budget is the room left, and that clamped +/// budget is what generation runs with and what `finish_reason` is judged against, +/// so a reply cut by the window reports "length", never "stop". +#[test] +fn st_budget_is_clamped_to_the_room_left_and_generation_stays_inside_it() { + let model = tiny_transformer(); + assert_eq!(st_context_budget(&model, 10, 8).ok(), Some(8)); + let budget = st_context_budget(&model, 60, 64).expect("60 of 64 fits"); + assert_eq!(budget, 4); + let prompt: Vec = (0..60).map(|i| 1 + (i % 11)).collect(); + let out = st_cpu_generate(&model, &prompt, budget, 0.0).expect("generates inside the window"); + let generated = out.len() - prompt.len(); + assert!(generated <= budget, "{generated} > budget {budget}"); + if generated == budget { + assert_eq!( + super::super::handlers::finish_reason_for(generated, budget), + "length" + ); + } +} diff --git a/crates/aprender-orchestrate/src/agent/code.rs b/crates/aprender-orchestrate/src/agent/code.rs index e57f6a8717..a7234034f6 100644 --- a/crates/aprender-orchestrate/src/agent/code.rs +++ b/crates/aprender-orchestrate/src/agent/code.rs @@ -1213,11 +1213,15 @@ fn run_single_prompt( let TurnPermit(()) = permit; let mut single_manifest = manifest.clone(); single_manifest.resources.max_iterations = single_manifest.resources.max_iterations.min(10); - // PMAT-197: Use compact system prompt for -p mode. - // The full CODE_SYSTEM_PROMPT (9-tool table + project context + CLAUDE.md) - // overwhelms Qwen3 1.7B causing loops. For -p mode, use a minimal - // prompt that lets the model answer directly. Tools still available if needed. - single_manifest.model.system_prompt = COMPACT_SYSTEM_PROMPT.to_string(); + // #3719: -p mode keeps the manifest's system prompt, which cmd_code has + // already scaled to the model (PMAT-198: COMPACT below 2B, the full tool + // table otherwise). PMAT-197 used to replace it here, for every model, + // with COMPACT_SYSTEM_PROMPT ("Answer the question. Be direct."), which + // names no tool and no format. The tools stayed registered but + // no model was told they existed: on Qwen3.5-4B (apr 0.69.0 cc3892acd, + // gx10, CUDA) `apr code -p` answered an edit-and-verify task with + // "Without seeing the actual code…" and made zero tool calls. PMAT-197's + // small-model concern is PMAT-198's <2B scaling, which still applies. // Note: context_window is set at driver launch time (build_default_manifest), // not here. See PMAT-197 fix in build_default_manifest. @@ -1433,7 +1437,7 @@ use super::code_envelope::{envelope, CodeOutcome}; // Prompts and exit codes extracted to code_prompts.rs use super::code_prompts::{ estimate_model_params_from_name, map_error_to_exit_code, scale_prompt_for_model, - CODE_SYSTEM_PROMPT, COMPACT_SYSTEM_PROMPT, + CODE_SYSTEM_PROMPT, }; #[cfg(test)] diff --git a/crates/aprender-orchestrate/src/agent/code_prompts.rs b/crates/aprender-orchestrate/src/agent/code_prompts.rs index 7aad2d094a..3d49087f4d 100644 --- a/crates/aprender-orchestrate/src/agent/code_prompts.rs +++ b/crates/aprender-orchestrate/src/agent/code_prompts.rs @@ -2,16 +2,29 @@ //! //! Extracted from code.rs to keep module under 500-line threshold. -/// Compact system prompt for -p mode (PMAT-197). +/// Compact system prompt for models under 2B parameters (PMAT-197/198). /// Minimal context to avoid overwhelming small models (Qwen3 1.7B). /// The full CODE_SYSTEM_PROMPT causes Qwen3 1.7B to loop on `` tags /// when combined with 9 tool JSON schemas consuming most of the context window. +/// +/// It names no tool and no `` format, so a model given only this +/// prompt cannot know it has tools. It was once forced on every `-p` run, +/// whatever the model size; that left `apr code -p` unable to edit a file on +/// any model (#3719). `scale_prompt_for_model` is its only caller now. pub(super) const COMPACT_SYSTEM_PROMPT: &str = "\ Answer the question. Be direct.\ "; /// System prompt — PMAT-168: optimized for 1.5B-7B with explicit tool table. /// +/// #3719: every example input here must carry the fields its tool REQUIRES; +/// `AprServeDriver` strips the JSON schemas, so this table is all a served +/// model sees. It taught `file_edit` with `old`/`new` (the tool requires +/// `old_string`/`new_string`) and `memory` with `key`/`value` (it requires +/// `content`), and Qwen3.5-4B called `file_edit` exactly as taught until the +/// loop guard ended the run. Pinned by +/// `falsify_3719_prompt_tool_examples_supply_every_required_field`. +/// /// 2026-05-20 update (V1_004 follow-up to paiml/claude-code-parity-apr M287): /// large coder-finetuned models (Qwen3-Coder-30B observed) emit Markdown /// `\u{60}\u{60}\u{60}rust` code blocks instead of `` JSON when given just @@ -35,11 +48,11 @@ You have 9 tools. To use one, emit a block: |------|---------|---------------| | file_read | Read a file | {\"path\": \"src/main.rs\"} | | file_write | Create/overwrite file | {\"path\": \"new.rs\", \"content\": \"fn main() {}\"} | -| file_edit | Replace text in file | {\"path\": \"src/lib.rs\", \"old\": \"foo\", \"new\": \"bar\"} | +| file_edit | Replace text in file | {\"path\": \"src/lib.rs\", \"old_string\": \"foo\", \"new_string\": \"bar\"} | | glob | Find files by pattern | {\"pattern\": \"src/**/*.rs\"} | | grep | Search file contents | {\"pattern\": \"TODO\", \"path\": \"src/\"} | | shell | Run a command | {\"command\": \"cargo test --lib\"} | -| memory | Remember/recall facts | {\"action\": \"remember\", \"key\": \"bug\", \"value\": \"off-by-one\"} | +| memory | Remember/recall facts | {\"action\": \"remember\", \"content\": \"bug: off-by-one\"} | | pmat_query | Search code by intent | {\"query\": \"error handling\", \"limit\": 5} | | rag | Search project docs | {\"query\": \"authentication flow\"} | @@ -56,7 +69,7 @@ Example 1 — read a file before editing: Example 2 — fix a one-line bug: -{\"name\": \"file_edit\", \"input\": {\"path\": \"src/lib.rs\", \"old\": \"return (i, j);\", \"new\": \"return (i.min(j), i.max(j));\"}} +{\"name\": \"file_edit\", \"input\": {\"path\": \"src/lib.rs\", \"old_string\": \"return (i, j);\", \"new_string\": \"return (i.min(j), i.max(j));\"}} Example 3 — verify with tests: @@ -137,7 +150,7 @@ pub(super) fn estimate_model_params_from_name(path: &std::path::Path) -> f64 { /// /// | Size | Prompt | Rationale | /// |------|--------|-----------| -/// | <2B | COMPACT | Avoids thinking loops, keeps tool format | +/// | <2B | COMPACT | Avoids thinking loops; names no tools | /// | 2-7B | MID | Tool names + format, no example JSON | /// | 7B+ | FULL | Full table with examples + guidelines | pub(super) fn scale_prompt_for_model(params_b: f64) -> String { diff --git a/crates/aprender-orchestrate/src/agent/code_tests.rs b/crates/aprender-orchestrate/src/agent/code_tests.rs index 4855d62658..248fe345e4 100644 --- a/crates/aprender-orchestrate/src/agent/code_tests.rs +++ b/crates/aprender-orchestrate/src/agent/code_tests.rs @@ -1463,3 +1463,170 @@ fn f3723_think_on_is_no_longer_refused_before_discovery() { ); assert!(msg.contains("apr serve is unavailable") && msg.contains("embedded fallback"), "{msg}"); } + +// --------------------------------------------------------------------------- +// #3719: a `-p` run must tell the model it has tools +// --------------------------------------------------------------------------- + +/// Records the system prompt of every request, and ends the turn at once. +/// +/// `run_agent_turn` sends `manifest.model.system_prompt` (plus any recalled +/// memories) as `request.system`, and `AprServeDriver` forwards it after +/// cutting only an `## Available Tools` section, which the coding prompt does +/// not carry. So what this driver records is what the served model reads. +struct SystemPromptRecorder { + systems: std::sync::Mutex>>, +} + +#[async_trait::async_trait] +impl LlmDriver for SystemPromptRecorder { + async fn complete( + &self, + request: crate::agent::driver::CompletionRequest, + ) -> Result { + self.systems.lock().expect("recorder lock").push(request.system.clone()); + Ok(crate::agent::driver::CompletionResponse { + text: "done".into(), + stop_reason: crate::agent::result::StopReason::EndTurn, + tool_calls: vec![], + usage: crate::agent::result::TokenUsage::default(), + }) + } + + fn context_window(&self) -> usize { + 32_768 + } + + fn privacy_tier(&self) -> PrivacyTier { + PrivacyTier::Sovereign + } +} + +/// FALSIFIER (#3719): a `-p` run sends the model the manifest's tool-bearing +/// system prompt. +/// +/// `run_single_prompt` used to replace it, for every model size, with +/// `COMPACT_SYSTEM_PROMPT` ("Answer the question. Be direct."), which names +/// no tool and no `` format. Measured on Qwen3.5-4B through +/// `apr serve` on CUDA: the model answered an edit-and-verify task with +/// "Without seeing the actual code…" and made no tool call, because nothing +/// it was sent said it could read or edit a file. RED on that code: the +/// recorded prompt is the 31-character COMPACT prompt. +#[test] +fn falsify_3719_single_prompt_run_tells_the_model_about_its_tools() { + let manifest = build_default_manifest(); + assert!( + manifest.model.system_prompt.contains(""), + "precondition: the default manifest carries the tool-call format" + ); + let tools = build_code_tools(&manifest); + let memory = crate::agent::memory::InMemorySubstrate::new(); + let driver = SystemPromptRecorder { systems: std::sync::Mutex::new(Vec::new()) }; + let mut budget = TurnBudget::new(1); + let permit = permit_single_prompt(&mut budget, /* non_interactive */ true) + .expect("one turn of budget must permit a -p run") + .expect("a -p run takes a permit"); + + let _ = run_single_prompt( + &manifest, + &driver, + &tools, + &memory, + "The unit test fails. Fix the bug with a one-line edit and run the test.", + None, + "text", + None, + permit, + ); + + let systems = driver.systems.lock().expect("recorder lock"); + let sent = systems + .first() + .and_then(|s| s.as_deref()) + .expect("the -p run must call the driver with a system prompt"); + assert_ne!( + sent, + crate::agent::code_prompts::COMPACT_SYSTEM_PROMPT, + "#3719: a -p run sent COMPACT_SYSTEM_PROMPT, which names no tool" + ); + assert!( + sent.contains(""), + "#3719: the model must be told the format in -p mode, got: {sent:?}" + ); + for tool in ["file_edit", "shell"] { + assert!( + sent.contains(tool), + "#3719: the -p system prompt must name `{tool}`, which the edit-and-verify task needs" + ); + } +} + +// --------------------------------------------------------------------------- +// #3719: every tool example the system prompt teaches must call the tool right +// --------------------------------------------------------------------------- + +/// Each `(tool name, example input)` the coding prompt shows the model: the +/// rows of its tool table and the JSON inside its `` examples. +fn prompt_tool_examples(prompt: &str) -> Vec<(String, serde_json::Value)> { + let mut out = Vec::new(); + for line in prompt.lines() { + let cells: Vec<&str> = line.split('|').map(str::trim).collect(); + if cells.len() >= 5 && cells[3].starts_with('{') { + if let Ok(input) = serde_json::from_str::(cells[3]) { + out.push((cells[1].to_string(), input)); + } + } + } + let mut rest = prompt; + while let Some(start) = rest.find("") { + let after = &rest[start + "".len()..]; + let Some(end) = after.find("") else { break }; + if let Ok(call) = serde_json::from_str::(after[..end].trim()) { + if let (Some(name), Some(input)) = (call["name"].as_str(), call.get("input")) { + out.push((name.to_string(), input.clone())); + } + } + rest = &after[end..]; + } + out +} + +/// FALSIFIER (#3719): the prompt's tool examples must supply every field the +/// registered tool REQUIRES. +/// +/// Measured on Qwen3.5-4B (lambda + gx10, CUDA, and on CPU through a logging +/// proxy): the model read both files, then called +/// `file_edit {"path": "stats.py", "old": …, "new": …}` exactly as the prompt's +/// table and Example 2 teach. The tool answered `missing required field +/// 'old_string'`, and the model repeated the call until the loop guard ended +/// the run with no answer. `AprServeDriver` strips the JSON schemas from the +/// prompt, so the table is all a served model sees. RED while the prompt says +/// `old`/`new`. +#[test] +fn falsify_3719_prompt_tool_examples_supply_every_required_field() { + let manifest = build_default_manifest(); + let tools = build_code_tools(&manifest); + let examples = prompt_tool_examples(CODE_SYSTEM_PROMPT); + assert!( + examples.iter().any(|(n, _)| n == "file_edit"), + "precondition: the prompt shows a file_edit example; parsed {examples:?}" + ); + let mut drift = Vec::new(); + for (name, input) in &examples { + let Some(tool) = tools.get(name) else { + continue; // a tool this build does not register (e.g. rag without its feature) + }; + let schema = tool.definition().input_schema; + for field in schema["required"].as_array().into_iter().flatten().filter_map(|f| f.as_str()) + { + if input.get(field).is_none() { + drift.push(format!("{name}: example {input} lacks required `{field}`")); + } + } + } + assert!( + drift.is_empty(), + "#3719: the prompt teaches tool calls the tools reject:\n{}", + drift.join("\n") + ); +} diff --git a/crates/aprender-orchestrate/src/agent/driver/realizar.rs b/crates/aprender-orchestrate/src/agent/driver/realizar.rs index 0400bc8045..b57f38dbbe 100644 --- a/crates/aprender-orchestrate/src/agent/driver/realizar.rs +++ b/crates/aprender-orchestrate/src/agent/driver/realizar.rs @@ -198,23 +198,25 @@ fn parse_tool_calls_envelope(text: &str) -> (String, Vec) { remaining.push_str(&cursor[..start]); let after_tag = &cursor[start + tag_len..]; - // Find closing tag and extract JSON - let (json_str, advance_past) = if is_markdown { + // Find closing tag and extract JSON. `delimited` is true only for a + // whose was found: the envelope, not a guess, + // says where the call ends. + let (json_str, advance_past, delimited) = if is_markdown { // Markdown: ```json\n...\n``` if let Some(end) = after_tag.find("```") { - (&after_tag[..end], &after_tag[end + "```".len()..]) + (&after_tag[..end], &after_tag[end + "```".len()..], false) } else { - (after_tag, "") + (after_tag, "", false) } } else if let Some(end) = after_tag.find("") { - (&after_tag[..end], &after_tag[end + "".len()..]) + (&after_tag[..end], &after_tag[end + "".len()..], true) } else { // PMAT-158: No closing tag — try parsing to end-of-string - (after_tag, "") + (after_tag, "", false) }; let json_str = json_str.trim(); - if let Ok(parsed) = serde_json::from_str::(json_str) { + if let Some(parsed) = parse_call_json(json_str, delimited) { // Must have "name" field to be a tool call (not just any JSON) if let Some(name) = parsed.get("name").and_then(|n| n.as_str()) { let name = name.to_string(); @@ -239,6 +241,74 @@ fn parse_tool_calls_envelope(text: &str) -> (String, Vec) { (remaining.trim().to_string(), tool_calls) } +/// #3719: recover a delimited `` whose JSON only lacks its closing +/// brackets. +/// +/// Measured on Qwen3.5-4B-Q4_K_M (gx10, CUDA): the model's final turn was +/// `{"name": "file_edit", "input": {"path": "stats.py", "old": "…", +/// "new": "…"}`. That is the right edit with the outer `}` missing, +/// so it failed to parse and was returned as the answer text, and the file +/// was never edited. The `` tag already marks where the call +/// ends, so the brackets still open there are closed in order. +/// +/// Conservative, like the salvage parser: `None` unless the scan ends outside +/// a string, nothing closes a bracket it did not open, at least one bracket is +/// still open, and the result is an object with a string `name` and an +/// explicit `input`. Any other malformation is left to fail as before. +fn repair_unclosed_tool_call(json_str: &str) -> Option { + let open = unclosed_brackets(json_str)?; + let mut repaired = json_str.to_string(); + repaired.extend(open.iter().rev()); + let parsed = serde_json::from_str::(&repaired).ok()?; + let obj = parsed.as_object()?; + obj.get("name")?.as_str().filter(|n| !n.is_empty())?; + obj.get("input")?; + info!("closed {} unclosed bracket(s) in a delimited (#3719)", open.len()); + Some(parsed) +} + +/// The JSON of one tool call: parsed as written, or, only for a call its +/// `` delimits, repaired by [`repair_unclosed_tool_call`]. +fn parse_call_json(json_str: &str, delimited: bool) -> Option { + let parsed = serde_json::from_str::(json_str).ok(); + if parsed.is_some() || !delimited { + return parsed; + } + repair_unclosed_tool_call(json_str) +} + +/// The closers still owed at the end of `json_str`, innermost last; `None` +/// when the scan ends inside a string, a closer does not match its opener, or +/// nothing is left open. +fn unclosed_brackets(json_str: &str) -> Option> { + let mut open: Vec = Vec::new(); + let (mut in_str, mut escaped) = (false, false); + for c in json_str.chars() { + if in_str { + (in_str, escaped) = string_step(c, escaped); + continue; + } + match c { + '"' => in_str = true, + '{' => open.push('}'), + '[' => open.push(']'), + '}' | ']' if open.pop() != Some(c) => return None, + _ => {} + } + } + (!in_str && !open.is_empty()).then_some(open) +} + +/// One character inside a JSON string: returns (still in the string, the next +/// character is escaped). +fn string_step(c: char, escaped: bool) -> (bool, bool) { + if escaped { + (true, false) + } else { + (c != '"', c == '\\') + } +} + /// CCPA-m296 salvage parser: recover a tool call the model emitted OUTSIDE the /// exact `` / ```json envelope, but in an unambiguous, recoverable /// shape. Two recoverable shapes are accepted, in priority order: @@ -606,4 +676,61 @@ not valid json assert_eq!(calls.len(), 1); assert_eq!(calls[0].id, "local-1", "envelope parser owns this, not salvage"); } + + // ── #3719: a delimited missing only its closing brackets ── + + /// FALSIFIER (#3719): the final turn Qwen3.5-4B-Q4_K_M produced on gx10 + /// (CUDA), verbatim. The edit is right and the outer `}` is missing, so the + /// call failed to parse, came back as the answer text, and `stats.py` was + /// never edited. RED before `repair_unclosed_tool_call`: zero calls. + #[test] + fn falsify_3719_delimited_tool_call_missing_outer_brace_is_executed() { + let input = "\n{\"name\": \"file_edit\", \"input\": {\"path\": \"stats.py\", \ + \"old\": \"return sum(values) / (len(values) - 1)\", \"new\": \"return \ + sum(values) / len(values)\"}\n"; + let (text, calls) = parse_tool_calls(input); + assert_eq!( + calls.len(), + 1, + "#3719: the delimited call must be executed, not returned as text" + ); + assert_eq!(calls[0].name, "file_edit"); + assert_eq!(calls[0].input["path"], "stats.py"); + assert_eq!(calls[0].input["old"], "return sum(values) / (len(values) - 1)"); + assert_eq!(calls[0].input["new"], "return sum(values) / len(values)"); + assert!(text.is_empty(), "the repaired call leaves no answer text, got {text:?}"); + } + + /// The repair closes brackets in the order they were opened, arrays too. + #[test] + fn repair_closes_nested_brackets_in_order() { + let input = + "{\"name\": \"shell\", \"input\": {\"args\": [\"a\", {\"b\": 1}"; + let (_text, calls) = parse_tool_calls(input); + assert_eq!(calls.len(), 1); + assert_eq!(calls[0].input["args"][1]["b"], 1); + } + + /// Everything else stays refused: the repair is not a general JSON fixer. + #[test] + fn repair_refuses_every_other_malformation() { + let refused = [ + // an unterminated string: where it ends is a guess + "{\"name\": \"file_edit\", \"input\": {\"path\": \"a", + // a bracket closed that was never opened this way + "{\"name\": \"file_edit\", \"input\": [}", + // no : nothing delimits the call + "{\"name\": \"file_edit\", \"input\": {\"path\": \"a\"}", + // no name + "{\"input\": {\"path\": \"a\"}", + // no input + "{\"name\": \"file_edit\", \"args\": {\"path\": \"a\"}", + // nothing left open: a different defect (a trailing comma) + "{\"name\": \"file_edit\", \"input\": {},}", + ]; + for input in refused { + let (_text, calls) = parse_tool_calls(input); + assert!(calls.is_empty(), "must not be repaired into a call: {input}"); + } + } } diff --git a/docs/audits/impl-PMAT-3719-receipt.md b/docs/audits/impl-PMAT-3719-receipt.md new file mode 100644 index 0000000000..cbc4a580a5 --- /dev/null +++ b/docs/audits/impl-PMAT-3719-receipt.md @@ -0,0 +1,69 @@ +# PMAT-3719 implementation receipt + +Issue #3719: `apr code` on Qwen3.5 Q4_K_M (4B, 9B), on CUDA, on lambda and gx10. Does it complete a scripted edit-and-verify task? +Branch `PMAT-3719-apr-code-qwen35-cuda`. Author: aprender-f8 (routed by the cop, aprender-04). Evidence: `evidence/apr-code-3719/`. + +## done_when 1: the gap, measured and posted + +- **Fixture:** `tests/fixtures/apr-code-edit-verify/`. `test_mean` fails, and the fix is one line in `stats.py`; `task.txt` asks for the fix and the test run. +- **Harness:** `scripts/apr_code_edit_verify.sh`. It runs `apr code -p --model M --project --output-format json --emit-trace ` under the fleet GPU lock (bounded flock, or `--gpu-q PRIO` with `GPUQ_WAIT`), choom 1000. +- **Judge:** `scripts/lib/apr_code_edit_verify.py` reads artifacts only: + - the working-copy diff; + - an independent test re-run; + - python shims on the agent's PATH (`--emit-trace` never records a tool call); + - the serve child's own output, captured through an `APR_BIN` wrapper. +- **Baseline gap** (`evidence/apr-code-3719/baseline-856009cc9/`), binary `apr 0.69.0 (856009cc9)` = main `52f43da71` + #3726 `c57c260bc` (the cop's rule: measure with #3726 in). All 4 cells FAIL "serve child did not load": `Model ready: 0 layers` · `gpu-layers: resolved=0 total=0`, then `CUDA optimized model ready`, then HTTP 500. That is #3571. Posted as https://github.com/paiml/aprender/issues/3719#issuecomment-5765960437. + +## done_when 2: every PASS cites the CUDA forward + +Final binary: `apr 0.69.0 (7b161d743)` = main `52f43da71` + #3726 + aprender-c7's #3571 step (2) `5a4a8e102` + this branch. Every PASS cell's `evidence` is the serve child's own line `Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens`, plus `gpu-layers: requested=all resolved=32 total=32 (backend=cuda)`, zero `[GPU->CPU FALLBACK]` lines, and a completion that came back. The judge reports `backend: cuda` only when all layers are resident on CUDA AND an answer returned. `CUDA optimized model ready` alone was measured to be printed for a 0-layer model, so it is not accepted. + +| Cell | v6 `7b161d743` | +|---|---| +| lambda · Qwen3.5-4B-Q4_K_M | PASS | +| lambda · Qwen3.5-9B-Q4_K_M | PASS | +| gx10 · Qwen3.5-4B-Q4_K_M | PASS | +| gx10 · Qwen3.5-9B-Q4_K_M | PASS | + +Honest limit: the route prints no per-request CUDA line, so the evidence is residency plus completions with no fallback marker. A per-request `prompt_tokens=N` line is promised by aprender-c7 after its receipt; until then rows are keyed `context: task`. + +## done_when 3: every FAIL was fixed here (falsifier RED on the pre-fix code) or points to its issue + +| Mechanism measured | Where | Fix | +|---|---|---| +| serve child did not load (0 layers → HTTP 500) | baseline, all 4 | #3571 (aprender-c7), under Refs | +| model never told it has tools: `-p` sent the system prompt "Answer the question. Be direct." to EVERY model (fake-serve capture) | gx10·4B on the #3571 route | `b2b89d69e`; `falsify_3719_single_prompt_run_tells_the_model_about_its_tools`, RED with the override restored | +| tool call not parsed: `` JSON missing only its outer `}` | gx10·4B v4 | `a35f7b8a0`; `falsify_3719_delimited_tool_call_missing_outer_brace_is_executed`, RED with the repair disabled | +| wrong edit: the prompt taught `file_edit` `old`/`new` (tool requires `old_string`/`new_string`) and `memory` `key`/`value` (requires `content`); the model repeated a rejected call until the loop guard ended the turn (logging-proxy capture) | lambda·4B v4/v5 | `ed714f055`; `falsify_3719_prompt_tool_examples_supply_every_required_field`, RED with 5 findings before | + +## done_when 4: the cells become release-matrix rows + +- The rows use the keys aprender-62 (#3712) and aprender-97 (#3715) agreed: + - the v2 cell fields; + - `thinking` read from the serve child's `chat template: … (thinking off|on)` line; + - `context` of `4k` only on a measured per-request `prompt_tokens ≥ 4096`, else `task`; + - `max_tokens` 1024. +- aprender-62's ruling: the ladder (#3712 row B2) calls this harness outside `apr_locked` with `--gpu-q 1`, once per derived code cell, and writes the ON row for a both-mode model as `verdict: error`, reason "apr code has no thinking toggle (#3723)". B2 waits on #3745. +- 62's condition: the harness must prove its `apr` call is locked, choom'd and bounded. `scripts/check_apr_code_edit_verify.sh` (dispatched by `guard_tree --no-cargo`) runs the real harness against a fake `apr` and a scratch lock, in all gate modes. Rows: free lock gives `held=yes oom=1000`; held lock gives a DECLINE within the bound. Mutants each turn it red: choom dropped, gpu-q bound dropped, `--caps` check dropped, a double lock. +- #3715's shape already derives verb `code` for every inventory model × host. A missing or non-pass row refuses the tag. + +## Checks +- `cargo test -p aprender-orchestrate --lib agent::`: 888 passed. `cargo clippy -p aprender-orchestrate --lib --tests -- -D warnings`: clean. `cargo fmt --check`: clean. +- `scripts/check_apr_code_edit_verify.sh`: 30 rows ok, 0 fail, mutation-checked. +- bashrs: 0 errors. `check_apr_bin_pinned`, `check_no_competing_harnesses`, `check_no_pipe_into_grep_q`, `check_guards_are_wired` and `check_roadmap_fragment_required`: pass. +- `scripts/guard_tree.sh --no-cargo` at `b9d78124c`: 77 checks, 1 failed (`check_complexity_ratchet.sh`: the tool-call repair grew `parse_tool_calls_envelope` and added an over-threshold function). Split in `e060bc114`; the ratchet then reads "PASS (D2): none new, none grown". + +## Out of scope, noted +- `apr code -p --output-format json` printed nothing on a driver error: #3775 (aprender-f8, separate branch). +- `--emit-trace` never records tool calls: filed by the cop for 0.70.0. + +## Re-measured after the orphan port (aprender-c1, 2026-09-27) + +The branch was orphaned on 2026-09-22 and carried onto main 761d6247d as `c1/3719-on-main` (merge bb79ae2c6; code_tests.rs resolved as main's file plus this branch's appended #3719 block). Binary `apr 0.69.3 (bb79ae2c66)`, `--features cuda`, sha256 prefix `ac4808c04195adbd`, built from that commit in a private target dir. RTX 4090 sm_89. Evidence: `evidence/apr-code-3719/v7-bb79ae2c6/`. + +| Cell | Verdict | Backend (from the serve child) | Agent ran the test | Test re-run here | +|------|---------|--------------------------------|--------------------|------------------| +| lambda · Qwen3.5-4B-Q4_K_M | PASS | cuda, 32 layers resident, no fallback | yes (shim log) | rc 0 | +| lambda · Qwen3.5-9B-Q4_K_M | PASS | cuda, 32 layers resident, no fallback | yes (shim log) | rc 0 | + +**Not re-measured:** the two gx10 cells. They still stand on the v6 binary `apr 0.69.0 (7b161d743)`. diff --git a/docs/audits/quorum-PMAT-3718,PMAT-3719.json b/docs/audits/quorum-PMAT-3718,PMAT-3719.json new file mode 100644 index 0000000000..e684543f5a --- /dev/null +++ b/docs/audits/quorum-PMAT-3718,PMAT-3719.json @@ -0,0 +1,338 @@ +{ + "ticket": "PMAT-3718,PMAT-3719", + "base": "origin/main", + "base_resolved": "origin/main", + "base_note": "no origin/origin/main exists; judged against the local ref", + "head": "3c07f85b5191ca3088bf439acd36c00cc003a4d2", + "diff_sha256": "1b0f8a395f2d50a3070156a08f5f612cf6818378d537a4cd93ac6c91cf5c5575", + "width": 3, + "executor": "agy", + "prompt_mode": "file", + "prompt_bytes": 259181, + "prompt_sha256": "68002576548f921db1108967558a26146eb93103dd6447f39e28431a1073c8b9", + "author": { + "model": "claude-opus-5-5", + "family": "claude", + "source": "flag" + }, + "agreed": true, + "lanes": [ + { + "lane": 1, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "Diff implements both tickets as scoped. PMAT-3718 (serve half): finish_reason_for(generated, max_tokens) replaces every hardcoded \"stop\" across SafeTensors chat/simple, wgpu streaming/non-streaming, and CUDA non-streaming/CPU-fallback handlers, and is judged consistently against the same clamped budget actually used for generation (verified max_tokens_clamped threading in handler_gpu_completion.rs:286-333). A new context_token_budget/context_length_exceeded_body pair (context_budget_3718.rs) refuses a prompt that fills the context window with 400 context_length_exceeded before prefill, wired into wgpu_chat_completion and the new st_context_budget (chat.rs, simple.rs), matching done_when 3. Verified code.rs:566-568 confirms the default manifest (used for ≥2B models, so Qwen3.5-4B/9B) still carries the full CODE_SYSTEM_PROMPT with the tool table — the receipt's central claim for PMAT-3719 (run_single_prompt no longer force-overwrites it with COMPACT_SYSTEM_PROMPT). The corrected tool-table field names (old_string/new_string, content) and the conservative bracket-repair for a delimited but truncated are each backed by a falsifier test asserted RED on the pre-fix code, with restrictive negative-case coverage (repair_refuses_every_other_malformation) so it doesn't become a general JSON fixer. Tests, evidence artifacts and the two new/updated quorum JSON receipts are internally consistent with the narrative. No test asserts the opposite of the ticket and no gate is weakened.", + "findings": [ + { + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, + "claim": "build_gpu_sse_stream's terminal [DONE] SSE event for the CUDA streaming chat path carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop — this function is untouched by the diff, so it's a pre-existing gap, not a regression, but it means #3718's guarantee is not uniform across every streaming transport the way it is for wgpu streaming.", + "grounding": "cited", + "fix": "Out of scope for this diff; worth a follow-up ticket if the CUDA streaming path should carry the same guarantee as wgpu streaming." + }, + { + "file": "crates/aprender-orchestrate/src/agent/code_prompts.rs", + "line": 154, + "claim": "scale_prompt_for_model's documented MID tier (2-7B) is dead code: the only call site (code.rs:566-568) gates the call itself on params_b < 2.0, so the MID and FULL branches of scale_prompt_for_model are never reached through cmd_code — the new comment describing the runtime behavior doesn't flag that the function it cites is partly unreachable.", + "grounding": "cited", + "fix": "Not a blocker for #3719 (the 4B/9B cells tested use the default full-prompt path, which is what the falsifiers check and the evidence shows worked); worth a follow-up to wire the MID branch into the call site or delete it." + } + ], + "raw_bytes": 6693, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "68002576548f921db1108967558a26146eb93103dd6447f39e28431a1073c8b9", + "trace": { + "input_sha256": "38cd717eb2b4870cd12f65894df527f4b6e4d1c64691345dd47c764745c16ec8", + "output_sha256": "4f476d7b40533a5e5f88373fcf647725c9ec591b68f225f8e58380f289312d04", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + }, + { + "lane": 2, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "The diff correctly implements the \"serve\" half of PMAT-3718 (finish_reason_for(generated, max_tokens) replacing every hardcoded \"stop\" across SafeTensors chat/simple, wgpu streaming/non-streaming, and CUDA/CPU-fallback handlers; a new context_token_budget/st_context_budget clamps generation and refuses over-long prompts with 400 context_length_exceeded before generation, wired before prefill in every touched path) and the tool-visibility/parsing fixes for PMAT-3719 in apr code -p mode (stops force-overwriting the manifest's tool-bearing system prompt with COMPACT_SYSTEM_PROMPT for ≥2B models, corrects the CODE_SYSTEM_PROMPT tool-call examples to match the tools' actual required fields, and adds a conservative bracket-repair for a missing only its closing brace). I verified against the actual worktree source (not just the diff) that build_default_manifest only applies COMPACT_SYSTEM_PROMPT below 2B params (code.rs:563-569) and that the clamped max_tokens correctly threads through both wgpu streaming and blocking completion paths (handlers.rs:381-406). New tests are meaningful falsifiers (asserted RED on pre-fix code per the receipt, e.g. falsify_3719_single_prompt_run_tells_the_model_about_its_tools, falsify_3719_delimited_tool_call_missing_outer_brace_is_executed, falsify_3719_prompt_tool_examples_supply_every_required_field, a_reply_cut_at_max_tokens_is_length_not_stop), none assert the opposite of the ticket, and the repair function is proven conservative by a negative case table (repair_refuses_every_other_malformation). No PMAT-3718 receipt file exists in this diff, but the receipt claims for this increment are only carried in the updated quorum-PMAT-3718.json narrative and are consistent with the actual code change; the ticket is explicitly 50% InProgress and this diff is only the serve-side half (the apr run --json half is noted, in the diff's own carried-forward audit data, as already landed in a prior commit on this branch). One pre-existing gap is honestly out of scope: the CUDA streaming terminal [DONE] SSE chunk (build_gpu_sse_stream, untouched by this diff) still carries no finish_reason at all, so a CUDA streaming client gets no length/stop signal — this is a real limitation but not introduced or misrepresented by this diff, and it was flagged in the diff's own carried-forward review data. Evidence/roadmap/script files for the #3719 harness were spot-checked (apr_code_edit_verify.sh) and are consistent with the receipt's described mechanism (independent test re-run, PATH shims, wrapper-captured serve-child output, GPU lock/choom/timeout bounds).", + "findings": [ + { + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "claim": "The CUDA streaming path's terminal SSE [DONE] chunk still reports no finish_reason at all (pre-existing, untouched by this diff), so a CUDA streaming client cannot distinguish a length-cut reply from a natural stop, unlike the wgpu streaming path which this diff does fix.", + "grounding": "cited", + "fix": "Follow-up ticket to extend the same finish_reason_for rule to the CUDA streaming terminal chunk, matching what this diff already did for wgpu streaming.", + "line": 236 + } + ], + "raw_bytes": 7272, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "68002576548f921db1108967558a26146eb93103dd6447f39e28431a1073c8b9", + "trace": { + "input_sha256": "5ebd2e9ea9d088b446ffc2ec0d17bbb4f09cd076736dc31c630c494a5f1438ce", + "output_sha256": "95e3ea8e5b83e3b8a240e661a10ceaaeac6fa2dd4e3025a457125101ba2e63b1", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [ + { + "model": "gemini-3.1-pro-high", + "why": "family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ] + }, + "repeat_of": 1 + }, + { + "lane": 3, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "Diff correctly implements both PMAT-3718 (finish_reason and context budgeting for serve endpoints) and PMAT-3719 (system prompt, tool call bracket repair, and prompt field names for apr code on Qwen3.5). All finish_reason determinations use clamped budgets, context overflows return 400 before generation, system prompt is no longer forced to COMPACT for all -p runs, tool examples use required field names, and bracket repair is strictly constrained to delimited calls. Receipt claims (gap, CUDA proof, fixes with falsifiers) are backed by diff. No gate weakening, no test asserting opposite of tickets, no scope creep.", + "findings": [], + "raw_bytes": 2344, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-haiku-4-5", + "model_measured": "claude-haiku-4-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "68002576548f921db1108967558a26146eb93103dd6447f39e28431a1073c8b9", + "trace": { + "input_sha256": "eeb71ab27d3ac084170b5b2e821d55af9b01812e009d652727111ee01b38855a", + "output_sha256": "8ae085634f5d0de8dc91ddf1bb332951e1e7eba5f4a3436da2878d2d7df6a8aa", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-haiku-4-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-haiku-4-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + } + ], + "dissent": [], + "dedup": [ + { + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, + "lanes_agreeing": [ + 1, + 2 + ], + "claims": [ + "The CUDA streaming path's terminal SSE [DONE] chunk still reports no finish_reason at all (pre-existing, untouched by this diff), so a CUDA streaming client cannot distinguish a length-cut reply from a natural stop, unlike the wgpu streaming path which this diff does fix.", + "build_gpu_sse_stream's terminal [DONE] SSE event for the CUDA streaming chat path carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop — this function is untouched by the diff, so it's a pre-existing gap, not a regression, but it means #3718's guarantee is not uniform across every streaming transport the way it is for wgpu streaming." + ] + }, + { + "file": "crates/aprender-orchestrate/src/agent/code_prompts.rs", + "line": 154, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "scale_prompt_for_model's documented MID tier (2-7B) is dead code: the only call site (code.rs:566-568) gates the call itself on params_b < 2.0, so the MID and FULL branches of scale_prompt_for_model are never reached through cmd_code — the new comment describing the runtime behavior doesn't flag that the function it cites is partly unreachable." + ] + } + ], + "uncovered": [], + "coverage_source": "lanes", + "partial": false, + "partial_reasons": [ + "lane models: only 2 distinct model ids across 3 lanes (claude-haiku-4-5, claude-sonnet-5) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)", + "lane 2: judged by claude-sonnet-5 after falling through gemini-3.1-pro-high (skipped: family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24) (PMAT-321)" + ], + "fallback": { + "same_family_width": 2, + "chain": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "gemini-3.1-pro-high", + "family": "gemini", + "disposition": "configured" + }, + { + "model": "claude-haiku-4-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "claude-opus-5-5", + "family": "claude", + "disposition": "excluded-self-review", + "why": "the author's own model id (claude-opus-5-5) — a model never reviews its own diff, at any width (R-15a identity bar)" + }, + { + "model": "qwen3.5", + "family": "qwen", + "disposition": "not-run", + "why": "no quorum.local_lane in the config — the aprender lane has no model to load" + } + ], + "precheck": [ + { + "family": "gemini", + "model": "gemini-3.1-pro-high", + "probe": 0, + "outcome": "policy", + "reason": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "prah": { + "source": "install-receipt", + "path": "/home/noah/.claude/skills/paiml-implement/bin/prah" + }, + "tier1": { + "agy_tier1_only": true, + "tier1": false, + "matched": [], + "paths": 170, + "patterns": [ + "^\\.github/workflows/", + "(^|/)release[^/]*\\.(ya?ml|sh|rs|toml)$", + "cuda|kernel|\\.cu$|\\.ptx$", + "(^|/)[^/]*(gate|guard)[^/]*\\.(sh|rs|py)$|^hooks/", + "security|secret|credential|(^|/)deny\\.toml$", + "^skills/quorum-review/|(^|/)(receipt-lint|roadmap-lint|release-lint|kind-gate|model-gate)[^/]*$|^crates/prah-lint/" + ], + "builtin": "^skills/quorum-review/|(^|/)skills/paiml-implement/config\\.json$|(^|/)(modellib|lane-reduce|lane-fallback|lane-group|agy-lane|cc-lane|receipt-lint|route|quota)\\.sh$|^crates/prah-lint/" + }, + "bucket": { + "ledger": "/home/noah/.local/state/paiml-implement/agy-bucket.jsonl", + "window_s": 18000, + "pace": "off", + "buckets": {}, + "closed": [] + } + }, + "auto_merge": { + "checked": true, + "was_armed": false, + "disarmed": false, + "note": "auto-merge not armed" + }, + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "cheap_seat": "claude-haiku-4-5", + "canary": { + "skipped": "brief 259181 bytes > 24576" + }, + "advisory_lane": { + "state": "not-run", + "counts": false, + "row": { + "NotRun": "ContextOverflow" + }, + "verdict": "unavailable", + "why": "the brief is 259181 bytes, over quorum.advisory_lane.max_brief_bytes=196608; a cut brief is a different question, so the lane was not launched", + "served_by": null, + "gpu_proof": null, + "apr": null, + "apr_binary_sha256": null, + "apr_tag": null, + "apr_commit": null, + "apr_build_why": null, + "model": null, + "model_sha256": null, + "rc": 0, + "wall_s": 0, + "collected_s": 0, + "budget_s": 170, + "budget_basis": "ledger: 2 x p95 85 s over 86 gx10-cuda Verdict rows", + "brief": { + "bytes": 259181, + "sent_bytes": 0, + "max_bytes": 196608 + }, + "trace": null, + "attempts": [], + "ledger": "/home/noah/.local/state/paiml-implement/advisory-ledger.jsonl", + "raw": "advisory.json", + "agrees_with_counted": null, + "counted": "PASS" + }, + "lint": { + "ok": true, + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5-5/claude same_family=3/2 degraded=same-family (policy: agy reserved for tier-1 diffs)" + } +} diff --git a/docs/audits/quorum-PMAT-3718.json b/docs/audits/quorum-PMAT-3718.json index 4c907d2ced..1b6e6301ce 100644 --- a/docs/audits/quorum-PMAT-3718.json +++ b/docs/audits/quorum-PMAT-3718.json @@ -1,16 +1,17 @@ { "ticket": "PMAT-3718", - "base": "225b2a9ab", - "base_resolved": "225b2a9ab", - "base_note": "no origin/225b2a9ab exists; judged against the local ref", - "head": "187b390e1dcfcd6c565d3f27aa779c06a60f8c82", - "diff_sha256": "20f9de301805687f8fb0e6c0853636df2b1aa308c0fbc03c28fe5ff343120b8f", + "base": "origin/main", + "base_resolved": "origin/main", + "base_note": "no origin/origin/main exists; judged against the local ref", + "head": "42eb448d14e6982756d6eca5a5d1a6cb7e1b4628", + "diff_sha256": "6180cac08f8236e7a69e0a5e5610e38f4c28f1449b0cf3c54f6b46f404790a75", "width": 3, "executor": "agy", "prompt_mode": "inline", - "prompt_bytes": 65794, + "prompt_bytes": 49675, + "prompt_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", "author": { - "model": "claude-opus-5", + "model": "claude-opus-5-5", "family": "claude", "source": "flag" }, @@ -20,50 +21,41 @@ "lane": 1, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff successfully implements PMAT-3718, correctly adding `prompt_tokens`, `completion_tokens`, and `finish_reason` to `apr run --json` and fixing `apr serve` endpoints to accurately report truncations. It resolves the silent stream cuts for clamped requests by judging the finish against the actual ran budget, maps unfittable prompts to a proper 400 error, and stops the APR CPU handler from echoing the prompt when no tokens are generated. The implementation conforms to all acceptance criteria and the provided receipt claims.", + "summary": "The diff correctly implements the \"serve\" half of PMAT-3718: a new finish_reason_for(generated, max_tokens) rule (\"length\" when generated >= the budget actually run with, else \"stop\") replaces every hardcoded \"stop\" in the SafeTensors chat handler, the wgpu non-streaming/streaming handlers, and the CUDA/CPU-fallback handlers. A new context_token_budget helper clamps max_tokens to the room left in the context window and refuses (400 context_length_exceeded) a prompt that fills it, wired into both the wgpu and SafeTensors paths before generation runs, matching done_when 3. Verified by reading (not running) the code: build_chat_response's finish_reason is judged against the clamped budget actually passed to st_cpu_generate (not the raw request), StCpuForward::context_length() exists and matches the comment's claim, and the tiny_transformer fixture used in the new 400-refusal test does have context_length 64 as asserted. No test asserts the opposite of the ticket, no gate is weakened, and I found no scope creep beyond what the ticket describes.", "findings": [ { - "claim": "Adds accurate token counts and finish_reason to CLI usage output using the actual engine count.", - "file": "crates/apr-cli/src/commands/inference_output.rs", - "grounding": "cited", - "line": 184 - }, - { - "claim": "Shadows max_tokens with clamped budget for correct length evaluation, and maps unfittable prompt errors to a 400 response.", - "file": "crates/aprender-serve/src/api/cuda_chat_backend.rs", - "grounding": "cited", - "line": 322 - }, - { - "claim": "Implements budget-aware finish reason determination correctly handling clamped runs and intrinsic stop token consumption.", - "file": "crates/aprender-serve/src/infer/run_report.rs", - "grounding": "cited", - "line": 39 - }, - { - "claim": "Fixes APR CPU fallback to yield empty instead of the prompt when nothing is generated, and uses `from_decode` to determine stop reason accurately.", - "file": "crates/apr-cli/src/commands/serve/handlers.rs", - "grounding": "cited", - "line": 1129 + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, + "claim": "Pre-existing, not introduced by this diff: build_gpu_sse_stream's terminal [DONE] SSE event carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop.", + "fix": "Out of scope for this diff (untouched file region) but worth a follow-up ticket if the CUDA streaming path is meant to satisfy the same #3718 guarantee as the wgpu streaming path, which does send a terminal chunk with finish_reason.", + "grounding": "cited" } ], - "raw_bytes": 4150, + "raw_bytes": 4281, "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-high", - "model_measured": "gemini-3.1-pro-high", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "cddc014e8538193133956da3e3252cb1fddeefe98d8a08941e3d14ac1d1f2859", + "output_sha256": "331b524226a250425a582f87c748927dca1bb4a7d0a7780fdc1aa0484df231d0", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-high", + "judged_by": "claude-sonnet-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-high", - "family": "gemini", + "model": "claude-sonnet-5", + "family": "claude", "outcome": "answered" } ], @@ -74,54 +66,76 @@ "lane": 2, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff strictly conforms to the acceptance criteria of PMAT-3718 and makes no unauthorized changes. `RunUsage` is implemented and emitted by `apr run --json` with `null` fallbacks. `FinishReason` correctly distinguishes cuts from stops via `from_decode`, fixing silent cuts in both CPU non-streaming and streaming `serve` paths. The context-clamped budgets are correctly used to evaluate the finish reason for the dense CPU loop, and over-long prompts now return a 400 error. The `a_fixed_text_prompt_counts_the_tokens_the_tokenizer_feeds` test adequately verifies the token count via the newly exposed `build_executable_pygmy_gguf_with` metadata hook. All receipt claims are backed by the code and tests.", + "summary": "Diff correctly implements PMAT-3718's serve-side requirement: a new finish_reason_for(generated, max_tokens) rule replaces hardcoded \"stop\" in SafeTensors chat/simple handlers, the CUDA non-streaming path, the CUDA→CPU fallback, and both wgpu streaming/non-streaming paths. A new context_token_budget clamps generation and refuses over-long prompts with 400 context_length_exceeded (SafeTensors via st_context_budget, wgpu via context_token_budget), applied before generation runs so the budget used for generation is the same one finish_reason is judged against. Traced the generated/max_tokens pairing through chat.rs, safetensors.rs, handler_gpu_completion.rs and handlers.rs — all internally consistent, no mismatched variables. New/updated tests match the intended behavior; no test asserts the opposite of the ticket. The \"apr run --json\" half of the ticket is already present on this branch (run.rs, run_tests_usage_3718.rs) from a prior commit, consistent with the 50%-InProgress status. handler_apr_cpu_completion.rs was left untouched but already applies the same length/stop rule inline (pre-existing), so it's not a gap. No gate weakening, no scope creep found.", "findings": [], - "raw_bytes": 2788, - "err_bytes": 100, + "raw_bytes": 3288, + "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-low", - "model_measured": "gemini-3.1-pro-low", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "e1f123825b8ff98955b709a499a6fe1b4eb2a8ba60ad9c2de24eae0772d2753b", + "output_sha256": "fa75e96bb14b90e3f0693b5f3029a2568d7339241a6ad15a4917b11df8a4157b", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-low", + "judged_by": "claude-sonnet-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-low", - "family": "gemini", + "model": "claude-sonnet-5", + "family": "claude", "outcome": "answered" } ], - "skipped": [] - } + "skipped": [ + { + "model": "gemini-3.1-pro-high", + "why": "family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ] + }, + "repeat_of": 1 }, { "lane": 3, "status": "SUCCESS", "verdict": "PASS", - "summary": "The diff perfectly implements the requested features from PMAT-3718. It adds `prompt_tokens`, `completion_tokens`, `finish_reason`, and `context_length` to `apr run --json` using the new `RunUsage` structure and the `run_inference_report` hook, returning `null` when a specific backend does not report these metrics. \nFor `apr serve`, it correctly calculates the `finish_reason` in `/v1/chat/completions` against the effective clamped budget rather than the requested one, preventing context-clamped responses from being falsely reported as `stop`. It correctly throws a 400 error (instead of truncating or throwing a 500 after the fact) when the prompt exceeds the context window. \nThe comprehensive test additions and updates enforce these behaviors properly, including the exact `a_fixed_text_prompt_counts_the_tokens_the_tokenizer_feeds` case row required by the ticket's done criteria.", + "summary": "Diff correctly implements PMAT-3718's serve half: replaces all hardcoded \"stop\" finish reasons with finish_reason_for(generated, max_tokens)→\"length\" when generated≥max_tokens, else \"stop\". Updates SafeTensors (chat/simple), wgpu (streaming/non-streaming), and CUDA (non-streaming/CPU-fallback) paths. Adds context budget validation that returns 400 context_length_exceeded before prefill when prompt fills window. Tests verify boundary cases and SafeTensors integration. No gates weakened, no unauthorized changes.", "findings": [], - "raw_bytes": 3040, + "raw_bytes": 1969, "err_bytes": 0, "envelope_status": "SUCCESS", "verdict_source": "structured_output", + "executor": "claude-code", "grounding_check": "parity", - "model": "gemini-3.1-pro-high", - "model_measured": "gemini-3.1-pro-high", - "model_source": "measured", - "family": "gemini", - "role": "independent", + "role": "counted", + "model": "claude-haiku-4-5", + "model_measured": "claude-haiku-4-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "8479b98ef43cfa5b3d4d5102ec14ff34913c3bc8ad188dd3473c54b5da582b65", + "trace": { + "input_sha256": "e090c837b5c14541d2a7899cc36a5f7bca39b8aef258c86b0ec3474fb65032a3", + "output_sha256": "7d5caa1c1ea251c78d226a68d9cee94e087b4d4f5ffa9535406729e8d2f29ff2", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, "fallback": { - "judged_by": "gemini-3.1-pro-high", + "judged_by": "claude-haiku-4-5", "exhausted": false, "attempts": [ { - "model": "gemini-3.1-pro-high", - "family": "gemini", + "model": "claude-haiku-4-5", + "family": "claude", "outcome": "answered" } ], @@ -132,43 +146,13 @@ "dissent": [], "dedup": [ { - "file": "crates/apr-cli/src/commands/inference_output.rs", - "line": 184, + "file": "crates/apr-cli/src/commands/serve/handler_gpu_completion.rs", + "line": 236, "lanes_agreeing": [ 1 ], "claims": [ - "Adds accurate token counts and finish_reason to CLI usage output using the actual engine count." - ] - }, - { - "file": "crates/apr-cli/src/commands/serve/handlers.rs", - "line": 1129, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Fixes APR CPU fallback to yield empty instead of the prompt when nothing is generated, and uses `from_decode` to determine stop reason accurately." - ] - }, - { - "file": "crates/aprender-serve/src/api/cuda_chat_backend.rs", - "line": 322, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Shadows max_tokens with clamped budget for correct length evaluation, and maps unfittable prompt errors to a 400 response." - ] - }, - { - "file": "crates/aprender-serve/src/infer/run_report.rs", - "line": 39, - "lanes_agreeing": [ - 1 - ], - "claims": [ - "Implements budget-aware finish reason determination correctly handling clamped runs and intrinsic stop token consumption." + "Pre-existing, not introduced by this diff: build_gpu_sse_stream's terminal [DONE] SSE event carries no finish_reason field at all (content chunks send finish_reason: null), so a CUDA streaming client never learns length vs stop." ] } ], @@ -176,66 +160,178 @@ "coverage_source": "lanes", "partial": false, "partial_reasons": [ - "lane models: only 2 distinct model ids across 3 lanes (gemini-3.1-pro-high, gemini-3.1-pro-low) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)" + "lane models: only 2 distinct model ids across 3 lanes (claude-haiku-4-5, claude-sonnet-5) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)", + "lane 2: judged by claude-sonnet-5 after falling through gemini-3.1-pro-high (skipped: family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24) (PMAT-321)" ], "fallback": { - "same_family_width": 1, + "same_family_width": 2, "chain": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, { "model": "gemini-3.1-pro-high", "family": "gemini", "disposition": "configured" }, { - "model": "gemini-3.1-pro-low", - "family": "gemini", - "disposition": "configured" + "model": "claude-haiku-4-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" }, { - "model": "gpt-oss-120b-medium", - "family": "openai", - "disposition": "fallback" + "model": "claude-opus-5-5", + "family": "claude", + "disposition": "excluded-self-review", + "why": "the author's own model id (claude-opus-5-5) — a model never reviews its own diff, at any width (R-15a identity bar)" }, { "model": "qwen3.5", "family": "qwen", "disposition": "not-run", "why": "no quorum.local_lane in the config — the aprender lane has no model to load" - }, - { - "model": "claude-opus-4-6-thinking", - "family": "claude", - "disposition": "width", - "why": "same family as the author: at most 1 lane, recorded role width, counted toward no floor (R-15a)" - }, - { - "model": "claude-sonnet-4-6", - "family": "claude", - "disposition": "width", - "why": "same family as the author: at most 1 lane, recorded role width, counted toward no floor (R-15a)" } ], "precheck": [ { "family": "gemini", "model": "gemini-3.1-pro-high", - "probe": 1, - "outcome": "live" + "probe": 0, + "outcome": "policy", + "reason": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" } ], + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, "prah": { "source": "install-receipt", "path": "/home/noah/.claude/skills/paiml-implement/bin/prah" + }, + "tier1": { + "agy_tier1_only": true, + "tier1": false, + "matched": [], + "paths": 8, + "patterns": [ + "^\\.github/workflows/", + "(^|/)release[^/]*\\.(ya?ml|sh|rs|toml)$", + "cuda|kernel|\\.cu$|\\.ptx$", + "(^|/)[^/]*(gate|guard)[^/]*\\.(sh|rs|py)$|^hooks/", + "security|secret|credential|(^|/)deny\\.toml$", + "^skills/quorum-review/|(^|/)(receipt-lint|roadmap-lint|release-lint|kind-gate|model-gate)[^/]*$|^crates/prah-lint/" + ], + "builtin": "^skills/quorum-review/|(^|/)skills/paiml-implement/config\\.json$|(^|/)(modellib|lane-reduce|lane-fallback|lane-group|agy-lane|cc-lane|receipt-lint|route|quota)\\.sh$|^crates/prah-lint/" + }, + "bucket": { + "ledger": "/home/noah/.local/state/paiml-implement/agy-bucket.jsonl", + "window_s": 18000, + "pace": "off", + "buckets": {}, + "closed": [] } }, "auto_merge": { - "checked": false, + "checked": true, "was_armed": false, "disarmed": false, - "note": "no --pr given: nothing to disarm" + "note": "auto-merge not armed" + }, + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "cheap_seat": "claude-haiku-4-5", + "canary": { + "skipped": "brief 49675 bytes > 24576" + }, + "advisory_lane": { + "state": "answered", + "counts": false, + "row": { + "Verdict": { + "verdict": "PASS", + "cell": "gx10-cuda", + "backend": "cuda" + } + }, + "verdict": "PASS", + "why": null, + "served_by": "gx10-cuda", + "gpu_proof": { + "used_gpu_probe": true, + "server_holds_gpu": true, + "used_gpu_source": "completions probe (chat omits used_gpu, aprender#4146)", + "trace_lines": [ + "3340528, 309 MiB" + ] + }, + "apr": "0.70.0", + "apr_binary_sha256": "6b2a7dc66c38adcca20b315beac70a1a31e7bd0af2a1cc5c2a7730ca6086d417", + "apr_tag": "v0.70.0-dev.3990584f0", + "apr_commit": "3990584f0bf5af6bca34ea9298df92177f3ef79a", + "apr_build_why": null, + "model": "/home/noah/data/models/Qwen3.5-4B-Q4_K_M.gguf", + "model_sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "rc": 0, + "wall_s": 15, + "collected_s": 821, + "budget_s": 128, + "budget_basis": "ledger: 2 x p95 64 s over 27 gx10-cuda Verdict rows", + "brief": { + "bytes": 49675, + "sent_bytes": 49844, + "max_bytes": 196608 + }, + "trace": { + "input_sha256": "568b8499c5f3b47f6bfce98326c10139208625a14c5b0a13ef052244dd0415c0", + "raw": "advisory.json" + }, + "attempts": [ + { + "host": "gx10", + "ok": true, + "wall_s": 15, + "used_gpu": true, + "ts": "2026-09-27T21:44:23.578Z", + "wall_ms": 15010, + "request_id": "01a0e4d3-8917-716f-b923-666910d9a5ea", + "load1": 12.20, + "cold": false + } + ], + "ledger": "/home/noah/.local/state/paiml-implement/advisory-ledger.jsonl", + "raw": "advisory.json", + "agrees_with_counted": true, + "counted": "PASS" }, "lint": { "ok": true, - "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5/claude" + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5-5/claude same_family=3/2 degraded=same-family (policy: agy reserved for tier-1 diffs)" } } diff --git a/docs/audits/quorum-PMAT-3719.json b/docs/audits/quorum-PMAT-3719.json new file mode 100644 index 0000000000..69253212f5 --- /dev/null +++ b/docs/audits/quorum-PMAT-3719.json @@ -0,0 +1,305 @@ +{ + "ticket": "PMAT-3719", + "base": "origin/main", + "base_resolved": "origin/main", + "base_note": "no origin/origin/main exists; judged against the local ref", + "head": "a96772bd0428f10928bc585a2a52167d6dd3caa2", + "diff_sha256": "7dde9c5676348b12b449f7d0d6db9db52e3366f2b0e19430a644fc5f03211bab", + "width": 3, + "executor": "agy", + "prompt_mode": "file", + "prompt_bytes": 199619, + "prompt_sha256": "07b318b67e45863b10cc34ed57def91f7a34fa9f77757c8508d04ffff735823c", + "author": { + "model": "claude-opus-5-5", + "family": "claude", + "source": "flag" + }, + "agreed": true, + "lanes": [ + { + "lane": 1, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "The diff fixes #3719 exactly as scoped: (1) `run_single_prompt` no longer force-overwrites the manifest's system prompt with COMPACT_SYSTEM_PROMPT for every model in `-p` mode, restoring tool visibility for ≥2B models (which get the default full CODE_SYSTEM_PROMPT via `build_default_manifest`/`assemble_system_prompt`, unaffected by the PMAT-198 <2B-only override at code.rs:565-570); (2) the tool-call examples in CODE_SYSTEM_PROMPT are corrected to the tools' actual required fields (`old_string`/`new_string`, `content`); (3) a conservative bracket-repair (`repair_unclosed_tool_call`/`unclosed_brackets`) recovers a `` missing only its closing brace, gated strictly to delimited calls and refusing every other malformation (verified by a negative case table). Each fix is backed by a new falsifier test that is asserted RED on the pre-fix code and is logically sound against the actual driver code read in this review. The evidence artifacts (cell.json, serve-child stdout/stderr, stats.diff, unittest.txt, trace.jsonl) for the two re-measured lambda cells are internally consistent with each other and with the receipt's claims (cuda, 32/32 layers resident, no fallback, rc 0, one-line edit, passing re-run). The judge script (`apr_code_edit_verify.py`) reads artifacts only, matches its documented mechanism-ordering, and its CUDA-vs-CPU/backend logic matches what's cited. The gx10 cells are honestly flagged as \"not re-measured\" rather than claimed. No test asserts the opposite of the ticket, no gate is weakened, and I found no unwrap/todo/ignored-test/vacuous-assertion patterns in the new code.", + "findings": [ + { + "file": "crates/aprender-orchestrate/src/agent/code_prompts.rs", + "line": 154, + "claim": "The MID_SYSTEM_PROMPT tier (2B–7B) documented in scale_prompt_for_model's table is dead code in production: the only call site (code.rs:565-570) gates the call itself on `params_b < 2.0`, so `scale_prompt_for_model`'s 2-7B and 7B+ branches are only ever exercised by direct unit tests, never through `cmd_code`. This diff's own new comment ('COMPACT below 2B, the full tool table otherwise') describes the actual (correct, for #3719's purposes) runtime behavior but glosses over the fact that the MID branch it documents is unreachable.", + "grounding": "cited", + "fix": "Not a blocker for #3719 (pre-existing, out of ticket scope, and the actual PASS evidence for the 4B/9B cells relies on the default full-prompt path, which does work) — worth a follow-up ticket to either wire MID into the call site or delete the dead branch.", + "failure_scenario": "None for this ticket: the 4B/9B models tested get the full CODE_SYSTEM_PROMPT regardless, which is what the falsifier tests check and what the evidence shows worked." + } + ], + "raw_bytes": 6386, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "07b318b67e45863b10cc34ed57def91f7a34fa9f77757c8508d04ffff735823c", + "trace": { + "input_sha256": "d7725f66b6c1f97bd0c9e05da2db4b43b8354ed0d8de45b0fa8d83e190030518", + "output_sha256": "9721bdbd7ffb466bebac3748cba47b309f969f3cf3ac9603468c28faa02dfe8e", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + }, + { + "lane": 2, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "Diff matches PMAT-3719's ticket and receipt. It restores the manifest's tool-bearing system prompt for `-p` mode (was unconditionally overwritten by COMPACT_SYSTEM_PROMPT, which named no tools), fixes the `file_edit`/`memory` tool-table examples to match the tools' actual required fields, and adds a conservative repair for a delimited `` missing only its closing brackets. Each of the three fixes carries a falsifier test that is shown RED on the pre-fix code (verified by reading the test bodies and the code they exercise), plus a large, well-structured harness/judge (apr_code_edit_verify.sh + apr_code_edit_verify.py) with an extensive case table (check_apr_code_edit_verify.sh) covering PASS, every FAIL mechanism, decline, and the harness's own locking/choom/gpu-q behavior including two named prior judge bugs (zero-layer CUDA-ready misread, trace-based test-run misread). The evidence files are consistent with the receipt's narrative across baseline → route → v6 → v7 stages. The receipt is honest about its limits: gx10 cells were not re-measured after the branch orphan-port and still cite the older v6 binary, and done_when 4 (release-matrix enforcement) is explicitly deferred to dependent tickets (#3712/#3715) rather than falsely claimed as done here. No gate weakening, no test asserting the opposite of the ticket, and no scope creep found.", + "findings": [], + "raw_bytes": 3660, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-sonnet-5", + "model_measured": "claude-sonnet-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "07b318b67e45863b10cc34ed57def91f7a34fa9f77757c8508d04ffff735823c", + "trace": { + "input_sha256": "39db290cc0b8ad5e30282ab54ea18f6fe485fc586262c68f52488bdf9ff84308", + "output_sha256": "3e270b9d70416f949dc65e2d79bcc7c32d8ba2a8842a54c7f9463681676efcb5", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-sonnet-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [ + { + "model": "gemini-3.1-pro-high", + "why": "family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ] + }, + "repeat_of": 1 + }, + { + "lane": 3, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "PMAT-3719 implementation is complete and correct. All three fixes (system prompt, tool call repair, prompt field names) are in place with falsifier tests. Evidence shows lambda cells PASS on v7 (bb79ae2c66) with backend: cuda and 32 layers resident. gx10 cells from v6 also PASS. Baseline demonstrates the gap. No gates weakened, no regressions detected.", + "findings": [], + "raw_bytes": 1626, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "executor": "claude-code", + "grounding_check": "parity", + "role": "counted", + "model": "claude-haiku-4-5", + "model_measured": "claude-haiku-4-5", + "model_source": "flag", + "family": "claude", + "brief_sha256": "07b318b67e45863b10cc34ed57def91f7a34fa9f77757c8508d04ffff735823c", + "trace": { + "input_sha256": "34bfaef107e051e89ef7fb79cbff922c9ce542c4df12f5bc6d81c6566d5aec54", + "output_sha256": "3e61d4cc24f44e8af5d512ef16b48df419bab908d09a24f35be28958cf27c413", + "store": null, + "store_why": "almacen not provisioned yet (infra-27): the blobs stay in the round's gitignored .lanes dir" + }, + "fallback": { + "judged_by": "claude-haiku-4-5", + "exhausted": false, + "attempts": [ + { + "model": "claude-haiku-4-5", + "family": "claude", + "outcome": "answered" + } + ], + "skipped": [] + } + } + ], + "dissent": [], + "dedup": [ + { + "file": "crates/aprender-orchestrate/src/agent/code_prompts.rs", + "line": 154, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "The MID_SYSTEM_PROMPT tier (2B–7B) documented in scale_prompt_for_model's table is dead code in production: the only call site (code.rs:565-570) gates the call itself on `params_b < 2.0`, so `scale_prompt_for_model`'s 2-7B and 7B+ branches are only ever exercised by direct unit tests, never through `cmd_code`. This diff's own new comment ('COMPACT below 2B, the full tool table otherwise') describes the actual (correct, for #3719's purposes) runtime behavior but glosses over the fact that the MID branch it documents is unreachable." + ] + } + ], + "uncovered": [], + "coverage_source": "lanes", + "partial": false, + "partial_reasons": [ + "lane models: only 2 distinct model ids across 3 lanes (claude-haiku-4-5, claude-sonnet-5) — lanes sharing an id are resamples, not independent reviewers (PMAT-125)", + "lane 2: judged by claude-sonnet-5 after falling through gemini-3.1-pro-high (skipped: family gemini withheld by policy: the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24) (PMAT-321)" + ], + "fallback": { + "same_family_width": 2, + "chain": [ + { + "model": "claude-sonnet-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "gemini-3.1-pro-high", + "family": "gemini", + "disposition": "configured" + }, + { + "model": "claude-haiku-4-5", + "family": "claude", + "disposition": "claude-code", + "why": "run by Claude Code on its own budget: a configured seat runs every round, a fallback step only when every non-Claude family is measured out (PMAT-360, operator standing rule)" + }, + { + "model": "claude-opus-5-5", + "family": "claude", + "disposition": "excluded-self-review", + "why": "the author's own model id (claude-opus-5-5) — a model never reviews its own diff, at any width (R-15a identity bar)" + }, + { + "model": "qwen3.5", + "family": "qwen", + "disposition": "not-run", + "why": "no quorum.local_lane in the config — the aprender lane has no model to load" + } + ], + "precheck": [ + { + "family": "gemini", + "model": "gemini-3.1-pro-high", + "probe": 0, + "outcome": "policy", + "reason": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "prah": { + "source": "install-receipt", + "path": "/home/noah/.claude/skills/paiml-implement/bin/prah" + }, + "tier1": { + "agy_tier1_only": true, + "tier1": false, + "matched": [], + "paths": 161, + "patterns": [ + "^\\.github/workflows/", + "(^|/)release[^/]*\\.(ya?ml|sh|rs|toml)$", + "cuda|kernel|\\.cu$|\\.ptx$", + "(^|/)[^/]*(gate|guard)[^/]*\\.(sh|rs|py)$|^hooks/", + "security|secret|credential|(^|/)deny\\.toml$", + "^skills/quorum-review/|(^|/)(receipt-lint|roadmap-lint|release-lint|kind-gate|model-gate)[^/]*$|^crates/prah-lint/" + ], + "builtin": "^skills/quorum-review/|(^|/)skills/paiml-implement/config\\.json$|(^|/)(modellib|lane-reduce|lane-fallback|lane-group|agy-lane|cc-lane|receipt-lint|route|quota)\\.sh$|^crates/prah-lint/" + }, + "bucket": { + "ledger": "/home/noah/.local/state/paiml-implement/agy-bucket.jsonl", + "window_s": 18000, + "pace": "off", + "buckets": {}, + "closed": [] + } + }, + "auto_merge": { + "checked": false, + "was_armed": false, + "disarmed": false, + "note": "no --pr given: nothing to disarm" + }, + "degraded": { + "reason": "same-family (policy: agy reserved for tier-1 diffs)", + "lanes": [ + 2 + ], + "families": [ + { + "family": "gemini", + "state": "policy", + "evidence": "the diff is not tier-1 (quorum.agy_tier1_only: none of its paths matches quorum.tier1_paths or the built-in policy paths), and agy is reserved for tier-1 diffs — cop ruling of 2026-09-24" + } + ], + "rule": "operator standing rule: if agy quota is ever gone, simply use claude code itself (PMAT-360)" + }, + "cheap_seat": "claude-haiku-4-5", + "advisory_lane": { + "state": "not-run", + "counts": false, + "row": { + "NotRun": "ContextOverflow" + }, + "verdict": "unavailable", + "why": "the brief is 199619 bytes, over quorum.advisory_lane.max_brief_bytes=24576; a cut brief is a different question, so the lane was not launched", + "served_by": null, + "gpu_proof": null, + "apr": null, + "model": null, + "model_sha256": null, + "rc": 0, + "wall_s": 0, + "collected_s": 0, + "budget_s": 128, + "budget_basis": "ledger: 2 x p95 64 s over 10 gx10-cuda Verdict rows", + "brief": { + "bytes": 199619, + "sent_bytes": 0, + "max_bytes": 24576 + }, + "trace": null, + "attempts": [], + "ledger": "/home/noah/.local/state/paiml-implement/advisory-ledger.jsonl", + "raw": "advisory.json", + "agrees_with_counted": null, + "counted": "PASS" + }, + "lint": { + "ok": true, + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5-5/claude same_family=3/2 degraded=same-family (policy: agy reserved for tier-1 diffs)" + } +} diff --git a/docs/roadmaps/entries/PMAT-3719.yaml b/docs/roadmaps/entries/PMAT-3719.yaml new file mode 100644 index 0000000000..fcb4b5ee9b --- /dev/null +++ b/docs/roadmaps/entries/PMAT-3719.yaml @@ -0,0 +1,21 @@ +- id: PMAT-3719 + github_issue: 3719 + item_type: task + title: '`apr code` on Qwen3.5 Q4_K_M (4B, 9B) on CUDA completes a scripted edit-and-verify task on lambda and gx10; UNMEASURED, and its driver runs through `apr serve --gpu` (#3571)' + status: in_progress + priority: critical + assigned_to: aprender-f8 + created: 2026-09-21T15:55:00Z + updated: 2026-09-21T17:40:00Z + spec: null + acceptance_criteria: + - 'Issue #3719 done_when 1, verbatim: "**Gap first, before any code.** Commit a fixture: a small project with one failing test whose fix is a one-line edit, and one task prompt that asks the agent to make the fix and run the test. Run `apr code -p --model --project --output-format json --emit-trace ` for M in {`Qwen3.5-4B-Q4_K_M`, `Qwen3.5-9B-Q4_K_M`} × {lambda, gx10}, with the 0.69.1 release-candidate binary (version and SHA cited). For each of the 4 cells, post on this issue: PASS or FAIL, the first failing mechanism named (serve child did not load / fell back to CPU / tool call not parsed / wrong edit / test not run / wrong final answer), and the stderr or trace line that shows the backend."' + - 'done_when 2, verbatim: "**CUDA engaged, proven.** Every PASS cell cites a line from the serve child or the trace showing the forward ran on CUDA. The requested flag does not count."' + - 'done_when 3, verbatim: "**Every FAIL cell** either gets its fix in this issue''s PR, with a falsifier that goes RED on the pre-fix binary, or points to the issue that carries the fix (for example #3571) under `Refs`."' + - 'done_when 4, verbatim: "**Enforced, not promised.** The four cells become rows (verb=`code`) of the release matrix that the #3715 shape refuses the tag on whenever a cell lacks a Pass receipt."' + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'Minted by aprender-c3 at the cop''s direction (aprender-04, 2026-09-21 ~15:45Z): "mint a row for it. Measure whether `apr code` drives a Qwen3.5 Q4_K_M (4B/9B) on CUDA through a scripted edit-and-verify task on both hosts, and post the gap before any code." The #3710 bar row it answers, verbatim: "**code**: `apr code` (the coding-agent verb) completes a scripted edit-and-verify task with the model as its backend". Known before measuring: `apr code` has no backend flag. Its default driver spawns `apr serve ... --gpu` (crates/aprender-orchestrate/src/agent/driver/apr_serve.rs:102), so its CUDA path is serve''s, and #3571 (serve cannot load Qwen3.5) sits on it. Reassigned to aprender-f8 by the cop (issue comment 2026-09-21T17:12:46Z, verbatim): "Owner: **aprender-f8** (routed by the cop, aprender-04). It moved from aprender-c3, which never started it and is carrying #3726/#3693/#3742. The PASS cells depend on #3571 step (2), owned by aprender-c7; the fixture and gap measurement (done_when 1) start now." Measurement binary, cop handoff verbatim: "take the MEASURED gap table on a binary that includes #3726 (its branch, or after the fold) and record which binary (version + sha) in every cell. A main-tip measurement mixes two defects." Row keys, ruled by aprender-62 (#3712) and aprender-97 (#3715): a code row carries the v2 cell fields plus thinking, context, prompt_tokens, max_tokens; thinking and backend are read from the serve child''s own output, never from a flag; context is `4k` only if a per-request prompt_tokens >= 4096 is measured (session_end.tokens_in is summed over turns and is not one); the thinking mode `apr code` cannot select is written as an explicit row, aprender-62 verbatim: `write the ON cell as verdict "error", reason "apr code has no thinking toggle (#3723)"`, which aprender-97 confirmed.' diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 06cc47fb2e..2a9f25952d 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -20710,6 +20710,27 @@ roadmap: labels: - kind:code notes: 'SCOPE. The three done_when rows quoted above. `prefill_ms` is deliberately NOT emitted by this row. The cop''s lane message asked for it, but it is not in #3718''s done_when, and it is owned elsewhere: aprender-a8 and aprender-3e agreed one definition (wall-clock over the forward passes on the post-template prompt, up to the first token''s logits; load, h2d, the F2 validate and tokenization excluded; null, never 0, where a path does not split prefill). It rides #3606''s StageTimings and 3e''s #3596 timer (`run_qwen35_generate_dispatch_timed`, not on main). Wiring it into this JSON is a follow-up once those land. RUN (done_when 1, 3). New `realizar::infer::run_inference_report(config) -> (InferenceResult, RunReport)`; `run_inference` is a thin wrapper over it. It is a separate entry point so that none of the ~75 `InferenceResult` literals change. `RunReport { finish_reason, context_length }` is None on a path that does not report it (APR, SafeTensors), never a default. The GGUF path computes it where the stop set and the budget are known. `FinishReason::from_decode(generated, stop, budget)`: a last token in the stop set is Stop; otherwise a run that filled its budget is Length; a shorter run ended on a stop token its loop consumed. This rule holds for both loop shapes, traced: the qwen35 CPU/GPU, MoE and wgpu loops PUSH the stop token, the dense CPU (`generate_with_cache`) and CUDA (`decode_blocking`) loops break BEFORE pushing, and every loop''s only other non-error exit is its budget. `decode_budget` clamps to `context_length - prompt` only on the dense CPU loop, the one loop that does (`effective_max_tokens`). `FinishReason` is declared once, in `infer::run_report` (not behind the `server` feature); `api::types::FinishReason` is now a re-export. apr-cli: `RunUsage { prompt_tokens, completion_tokens, finish_reason, context_length }` on `RunResult`, emitted by `build_final_json`, so it appears in both `--json` and the `--stream` final event. `prompt_tokens` = `input_token_count`, the post-chat-template count including BOS that `prepare_tokens` fed the model. `completion_tokens` is the engine count, never the word-count stand-in `tokens_generated` falls back to. No path truncates a prompt (traced: `prepare_tokens` has no truncation). An over-long prompt is REFUSED (ContextLimitExceeded -> error exit), which is done_when 3''s "or an error"; the completion-side cut is now `finish_reason: length`, where before it was silent. SERVE (done_when 2, 3). `/v1/chat/completions` `usage` was already OpenAI-shaped with the post-template count on the realizar backends (`tokenize_chat_prompt`: template, encode, `len`). Three silent cuts are fixed. (a) The dense CPU backend (`try_quantized_backend`) judged finish against the REQUESTED max_tokens after `effective_max_tokens` had clamped the loop to the context room, so a context-clamped cut said "stop"; it is now judged against the budget the loop ran with, both streaming and not. (b) A prompt that cannot fit was a 500 (non-stream) or a 200 SSE carrying an error event that still ended "stop". It is now a 400 before generation on both, via the existing `generation_error_status`, whose doc already said callers map it to 400. (c) The APR CPU fallback (apr-cli handler_apr_cpu_completion.rs) hardcoded "stop" while capping max_tokens at 4096. It now uses `from_decode` against that cap with the loop''s real stop set (token 0, pushed; `is_eos_token`). Its `run_apr_cpu_inference` also returned the PROMPT as the completion when nothing was generated (`else { &output_tokens[..] }`); that now yields empty. MEASURED on lambda, CPU, Qwen3.5-0.8B-Q4_K_M, pinned `apr 0.69.0 (3fadee114)` via scripts/apr_bin.sh. `apr run -i <"What is 2+2? Answer with one number."> --chat --json`: prompt_tokens 24, context_length 262144. The same templated string through llama.cpp 41fc758 `llama-tokenize --ids` gives 24 ids, identical id-for-id to apr''s `--verbose` encode (with and without llama.cpp''s BOS policy). That is done_when 1''s case row against an independent tokenizer. max-tokens 4 and 1024: completion 2, finish stop (the model ends on EOS). A 20-line-poem prompt at max-tokens 6: completion 6, finish length. DONE_WHEN 1 CASE ROW, automated (added after the round-2 quorum FAIL, which correctly said a manual measurement is not a case row): infer::run_report::a_fixed_text_prompt_counts_the_tokens_the_tokenizer_feeds builds the pygmy GGUF with a SentencePiece vocabulary (` ▁Hello ▁world !` + filler; bos 1, eos 2) through the new `test_factory::build_executable_pygmy_gguf_with` metadata hook, runs the TEXT prompt "Hello world!" through `run_inference_report` (the `apr run --prompt` path: prepare_tokens_gguf, the GGUF tokenizer, BOS policy) and asserts the fed ids are [1, 3, 4, 5] and input_token_count (which becomes prompt_tokens) is 4. The expectation is derived by hand from the vocabulary, not from apr''s encoder. Mutant: prepare_tokens_gguf reports tokens.len()-1 -> the row goes RED. The post-template count on a real model is the llama.cpp row above (24 == 24, id for id). TESTS. infer::run_report: a case table over both loop shapes, the clamped-cut pin, budget arithmetic, and four rows through `run_inference_report` on the pygmy GGUF (context 32): a context-clamped cut is Length with 28 = 32 - 4 tokens, a max_tokens cut is Length, a stop token under budget is Stop (0 tokens, dense loop consumed it), and an over-long prompt is refused naming both lengths. api::tests::usage_finish_3718 runs through the real router on a quantized AppState: clamped cut -> length with usage prompt+completion == context, clamped stream -> terminal chunk length, prompt > context -> 400 on stream and non-stream, stop under budget -> stop (both), unclamped -> cut at the request budget. apr-cli run_tests_usage_3718: JSON carries the four fields, null (not 0 or "stop") when unreported, stream final event carries them. MUTANTS, each run and each RED: serve clamp reverted (2 red), 400 reverted to the old path (1), run-report clamp off (1), from_generation always Length (4), never Length (7). SUITES at 3fadee114 + rustfmt: `cargo test -p aprender-serve --lib` 15913 passed; `cargo test -p apr-cli --lib` 7301 passed; `cargo test -p aprender-contracts --lib` 1684 passed; clippy -D warnings on aprender-serve and apr-cli clean; fmt clean. KNOWN, NOT CHANGED. completion_tokens counts the pushed stop token on the qwen35/MoE/wgpu loops and not on the dense loops, because that is what each loop returns (measured: qwen35 `tokens: [17, 248046]`, completion 2). The CUDA backends were not exercised on hardware in this row: the report logic is backend-agnostic and clamps only on the dense CPU loop. Streaming chat does not apply stop STRINGS on the true-streaming path; that is pre-existing and not a length cut. Found while measuring and routed separately: `--prompt P --chat` double-templates (24 -> 56 prompt_tokens), which is #3743 = #3672 in batch-1.' +- id: PMAT-3719 + github_issue: 3719 + item_type: task + title: '`apr code` on Qwen3.5 Q4_K_M (4B, 9B) on CUDA completes a scripted edit-and-verify task on lambda and gx10; UNMEASURED, and its driver runs through `apr serve --gpu` (#3571)' + status: in_progress + priority: critical + assigned_to: aprender-f8 + created: 2026-09-21T15:55:00Z + updated: 2026-09-21T17:40:00Z + spec: null + acceptance_criteria: + - 'Issue #3719 done_when 1, verbatim: "**Gap first, before any code.** Commit a fixture: a small project with one failing test whose fix is a one-line edit, and one task prompt that asks the agent to make the fix and run the test. Run `apr code -p --model --project --output-format json --emit-trace ` for M in {`Qwen3.5-4B-Q4_K_M`, `Qwen3.5-9B-Q4_K_M`} × {lambda, gx10}, with the 0.69.1 release-candidate binary (version and SHA cited). For each of the 4 cells, post on this issue: PASS or FAIL, the first failing mechanism named (serve child did not load / fell back to CPU / tool call not parsed / wrong edit / test not run / wrong final answer), and the stderr or trace line that shows the backend."' + - 'done_when 2, verbatim: "**CUDA engaged, proven.** Every PASS cell cites a line from the serve child or the trace showing the forward ran on CUDA. The requested flag does not count."' + - 'done_when 3, verbatim: "**Every FAIL cell** either gets its fix in this issue''s PR, with a falsifier that goes RED on the pre-fix binary, or points to the issue that carries the fix (for example #3571) under `Refs`."' + - 'done_when 4, verbatim: "**Enforced, not promised.** The four cells become rows (verb=`code`) of the release matrix that the #3715 shape refuses the tag on whenever a cell lacks a Pass receipt."' + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'Minted by aprender-c3 at the cop''s direction (aprender-04, 2026-09-21 ~15:45Z): "mint a row for it. Measure whether `apr code` drives a Qwen3.5 Q4_K_M (4B/9B) on CUDA through a scripted edit-and-verify task on both hosts, and post the gap before any code." The #3710 bar row it answers, verbatim: "**code**: `apr code` (the coding-agent verb) completes a scripted edit-and-verify task with the model as its backend". Known before measuring: `apr code` has no backend flag. Its default driver spawns `apr serve ... --gpu` (crates/aprender-orchestrate/src/agent/driver/apr_serve.rs:102), so its CUDA path is serve''s, and #3571 (serve cannot load Qwen3.5) sits on it. Reassigned to aprender-f8 by the cop (issue comment 2026-09-21T17:12:46Z, verbatim): "Owner: **aprender-f8** (routed by the cop, aprender-04). It moved from aprender-c3, which never started it and is carrying #3726/#3693/#3742. The PASS cells depend on #3571 step (2), owned by aprender-c7; the fixture and gap measurement (done_when 1) start now." Measurement binary, cop handoff verbatim: "take the MEASURED gap table on a binary that includes #3726 (its branch, or after the fold) and record which binary (version + sha) in every cell. A main-tip measurement mixes two defects." Row keys, ruled by aprender-62 (#3712) and aprender-97 (#3715): a code row carries the v2 cell fields plus thinking, context, prompt_tokens, max_tokens; thinking and backend are read from the serve child''s own output, never from a flag; context is `4k` only if a per-request prompt_tokens >= 4096 is measured (session_end.tokens_in is summed over turns and is not one); the thinking mode `apr code` cannot select is written as an explicit row, aprender-62 verbatim: `write the ON cell as verdict "error", reason "apr code has no thinking toggle (#3723)"`, which aprender-97 confirmed.' - id: PMAT-3724 github_issue: 3724 item_type: task diff --git a/evidence/apr-code-3719/README.md b/evidence/apr-code-3719/README.md new file mode 100644 index 0000000000..543306d5a9 --- /dev/null +++ b/evidence/apr-code-3719/README.md @@ -0,0 +1,77 @@ +# #3719: `apr code` edit-and-verify on Qwen3.5 Q4_K_M, CUDA, lambda + gx10 + +Each cell is one run of `scripts/apr_code_edit_verify.sh --model --host --out ` +against the fixture `tests/fixtures/apr-code-edit-verify` (one failing test, a one-line fix). +`cell.json` is the verdict of `scripts/lib/apr_code_edit_verify.py`, re-run over each directory's +artifacts with the judge at `005d53f79` (md5 `72fab3e4cd98b8f17ed1ee18c2a4b032` on both hosts). +The judge reads only artifacts, never the agent's claims: + +- the working-copy diff; +- an independent test re-run; +- the python-shim log; +- the serve child's own stdout and stderr. + +Case table: `scripts/check_apr_code_edit_verify.sh`. + +## baseline-856009cc9: the gap (#3719 done_when 1) + +Binary `apr 0.69.0 (856009cc9)`, built with `--features cuda`. It is a local merge, never pushed: +origin/main `52f43da71` + `PMAT-3726-canonical-bpe` `c57c260bc` (the cop: measure on a binary that +includes #3726) + the harness `e63e90849`. No 0.69.1 tag or version bump existed at measurement time. + +| Model | lambda (RTX 4090) | gx10 (GB10) | +|---|---|---| +| Qwen3.5-4B-Q4_K_M | FAIL: serve child did not load | FAIL: serve child did not load | +| Qwen3.5-9B-Q4_K_M | FAIL: serve child did not load | FAIL: serve child did not load | + +All four fail the same way. The serve child printed `Model ready: 0 layers` and +`gpu-layers: requested=all resolved=0 total=0 (backend=cuda)`. It then printed +`CUDA optimized model ready` anyway, and answered the first completion with HTTP 500: +`Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for +GPU-resident path`. Fix: #3571 (step 1 refuses the 0-layer load; step 2 adds the Qwen3.5 route). + +No CUDA forward ran in any cell. `CUDA optimized model ready` is a load-time line, and it is printed +for a model with no layers. + +## route-cc3892acd: the next mechanism + +Binary `apr 0.69.0 (cc3892acd)` = the baseline + aprender-c7's #3571 step (2), branch +`PMAT-3571-serve-qwen35-session-v2` @ `5a4a8e102` (pre-quorum-receipt), + the harness `005d53f79`. + +gx10 / Qwen3.5-4B: + +- **Serve child:** loaded, `Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU` · + `chat template: Qwen3NoThink (thinking off)` · `gpu-layers: requested=all resolved=32 total=32 + (backend=cuda)`. The model then answered coherently. +- **Agent:** made no tool call and left `stats.py` unchanged. Its answer began "Without seeing the + actual code, I'll assume…". The run was `num_turns 1` and `tokens_in 81`. +- **Cause:** `fake-serve-request-capture.jsonl` shows what `apr code -p` sent. The system message + was `Answer the question. Be direct.` (31 chars): `run_single_prompt` replaced the coding prompt + with `COMPACT_SYSTEM_PROMPT` for every model, so no model was told it had tools. +- **Fix:** in this PR (`b2b89d69e`), with falsifier + `falsify_3719_single_prompt_run_tells_the_model_about_its_tools`. It is RED with the old + override restored. + +## v6-7b161d743: all four cells PASS (#3719 done_when 1–3) + +Binary `apr 0.69.0 (7b161d743)`, `--features cuda`, the same SHA on both hosts. It is a local merge, never pushed: the baseline + aprender-c7's #3571 step (2) `5a4a8e102` + this branch through `ed714f055`. The harness ran through `gpu-q --prio 1` (`GPUQ_WAIT` bound). + +| Model | lambda (RTX 4090) | gx10 (GB10) | +|---|---|---| +| Qwen3.5-4B-Q4_K_M | **PASS** (5 turns, 81 s) | **PASS** | +| Qwen3.5-9B-Q4_K_M | **PASS** | **PASS** (4 turns, 214 s) | + +Each PASS cell carries: + +- **the serve child's own lines:** `Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens` · `chat template: Qwen3NoThink (thinking off)` · `gpu-layers: requested=all resolved=32 total=32 (backend=cuda)`, and zero `[GPU->CPU FALLBACK]` lines; +- **the edit:** exactly `- return sum(values) / (len(values) - 1)` → `+ return sum(values) / len(values)`, with `test_stats.py` untouched and no files added; +- **the test run:** the agent's own `python3 -m unittest test_stats -v` (shim log), and the independent re-run `OK`; +- **the answer:** it names the off-by-one and reports the passing tests. + +The fixes between route-cc3892acd and v6, each with a falsifier that was RED on the pre-fix code: + +1. `b2b89d69e`: `-p` stopped replacing the coding prompt with "Answer the question. Be direct.". +2. `a35f7b8a0`: a delimited `` missing only its closing brackets is executed. +3. `ed714f055`: the prompt teaches `file_edit` `old_string`/`new_string` and `memory` `content`, the fields the tools require. Before, Qwen3.5-4B repeated a rejected `old`/`new` call until the loop guard ended the turn; a logging proxy in front of the serve child showed it. + +Rows are keyed `thinking: off` (read from the child) and `context: task` (the route prints no per-request prompt size yet). The ON row that `apr code` cannot produce is written by the #3712 ladder as `verdict: error`. diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/apr-version.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/apr-version.txt new file mode 100644 index 0000000000..f571b481b8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (856009cc9) diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/cell.json b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/cell.json new file mode 100644 index 0000000000..aa5f50ed0c --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/cell.json @@ -0,0 +1,27 @@ +{ + "verb": "code", + "host": "gx10", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.0 (856009cc9)", + "hostname": "gx10-a5b5", + "started": "2026-09-21T17:38:41Z", + "finished": "2026-09-21T18:18:14Z", + "rc": 1, + "test_rc": 1, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "unknown", + "fallback": false, + "evidence": "Model ready: 0 layers, vocab_size=248320, hidden_dim=2560\ngpu-layers: requested=all resolved=0 total=0 (backend=cuda)", + "model_layers": 0, + "answer_chars": 0, + "thinking": "unknown", + "thinking_raw": "", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [], + "verdict": "fail", + "mechanism": "serve child did not load", + "reason": "the child reported ready with 0 layers; first request: HTTP 500 {\"error\":\"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path\"}" +} diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/finished.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/finished.txt new file mode 100644 index 0000000000..6b1b30f41d --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/finished.txt @@ -0,0 +1 @@ +2026-09-21T18:18:14Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/hostname.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/hostname.txt new file mode 100644 index 0000000000..c9c46b28f5 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/hostname.txt @@ -0,0 +1 @@ +gx10-a5b5 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/lock-acquired b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/lock-acquired new file mode 100644 index 0000000000..2e585bcfa2 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T18:18:05Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/model.sha256 b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stderr b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stderr new file mode 100644 index 0000000000..905cfd628e --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stderr @@ -0,0 +1,33 @@ +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 3 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 3 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 7 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 7 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-129] Early kernel preload: 44 modules compiled +[PMAT-044] Batch scheduler started: max_batch=32, window=0ms +[PMAT-044] Batch m=1 done in 0.6ms (1558596.2 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (54489973.8 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (50362610.8 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (60562015.5 tok/s/slot) diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stdout b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stdout new file mode 100644 index 0000000000..4fb6b22d9c --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/serve-child.stdout @@ -0,0 +1,33 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:20070 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Building quantized inference model... +Model ready: 0 layers, vocab_size=248320, hidden_dim=2560 +gpu-layers: requested=all resolved=0 total=0 (backend=cuda) +Enabling optimized CUDA acceleration (PAR-111)... + Max sequence length: 4096 + Initializing GPU on device 0... + Pre-uploaded 497 MB weights to GPU +CUDA optimized model ready + CONTINUOUS BATCHING: max_batch=32, window=0ms (PMAT-044) + +CUDA-optimized server listening on http://127.0.0.1:20070 + +Endpoints: + GET /health - Health check + GET /metrics - Prometheus metrics + POST /generate - Text generation + POST /v1/completions - OpenAI-compatible + POST /v1/chat/completions - Chat completions + +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/started.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/started.txt new file mode 100644 index 0000000000..28464f4827 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/started.txt @@ -0,0 +1 @@ +2026-09-21T17:38:41Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/stderr.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/stderr.txt new file mode 100644 index 0000000000..2f812022bc --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/stderr.txt @@ -0,0 +1,3 @@ +Launched apr serve on port 20070 (pid 702689) +apr serve ready (2.0s) +Error: driver error: network error: apr serve HTTP 500: {"error":"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path"} diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/stdout.json b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/stdout.json new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/unittest.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/unittest.txt new file mode 100644 index 0000000000..63518b4678 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-4B/unittest.txt @@ -0,0 +1,13 @@ +F. +====================================================================== +FAIL: test_mean (test_stats.TestStats.test_mean) +---------------------------------------------------------------------- +Traceback (most recent call last): + File "/home/noah/wt-3719-logs/cells/gx10-4B/project/test_stats.py", line 8, in test_mean + self.assertEqual(mean([2, 4, 6]), 4) +AssertionError: 6.0 != 4 + +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +FAILED (failures=1) diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/apr-version.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/apr-version.txt new file mode 100644 index 0000000000..f571b481b8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (856009cc9) diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/cell.json b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/cell.json new file mode 100644 index 0000000000..946d723f02 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/cell.json @@ -0,0 +1,27 @@ +{ + "verb": "code", + "host": "gx10", + "file": "Qwen3.5-9B-Q4_K_M.gguf", + "sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", + "apr_version": "apr 0.69.0 (856009cc9)", + "hostname": "gx10-a5b5", + "started": "2026-09-21T18:18:17Z", + "finished": "2026-09-21T19:02:47Z", + "rc": 1, + "test_rc": 1, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "unknown", + "fallback": false, + "evidence": "Model ready: 0 layers, vocab_size=248320, hidden_dim=4096\ngpu-layers: requested=all resolved=0 total=0 (backend=cuda)", + "model_layers": 0, + "answer_chars": 0, + "thinking": "unknown", + "thinking_raw": "", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [], + "verdict": "fail", + "mechanism": "serve child did not load", + "reason": "the child reported ready with 0 layers; first request: HTTP 500 {\"error\":\"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path\"}" +} diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/finished.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/finished.txt new file mode 100644 index 0000000000..38c1a1fac9 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/finished.txt @@ -0,0 +1 @@ +2026-09-21T19:02:47Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/hostname.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/hostname.txt new file mode 100644 index 0000000000..c9c46b28f5 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/hostname.txt @@ -0,0 +1 @@ +gx10-a5b5 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/lock-acquired b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/lock-acquired new file mode 100644 index 0000000000..4fccc316ec --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T19:02:24Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/model.sha256 b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/model.sha256 new file mode 100644 index 0000000000..a25e916189 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/model.sha256 @@ -0,0 +1 @@ +03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stderr b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stderr new file mode 100644 index 0000000000..7da5e5d4a3 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stderr @@ -0,0 +1,33 @@ +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 3 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 3 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 7 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 7 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-129] Early kernel preload: 44 modules compiled +[PMAT-044] Batch scheduler started: max_batch=32, window=0ms +[PMAT-044] Batch m=1 done in 0.2ms (4580789.1 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (58247903.1 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (58520599.3 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (58685446.0 tok/s/slot) diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stdout b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stdout new file mode 100644 index 0000000000..1341907cbc --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/serve-child.stdout @@ -0,0 +1,33 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-9B-Q4_K_M.gguf +Binding: 127.0.0.1:19991 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 427 tensors, 46 metadata entries +Building quantized inference model... +Model ready: 0 layers, vocab_size=248320, hidden_dim=4096 +gpu-layers: requested=all resolved=0 total=0 (backend=cuda) +Enabling optimized CUDA acceleration (PAR-111)... + Max sequence length: 4096 + Initializing GPU on device 0... + Pre-uploaded 795 MB weights to GPU +CUDA optimized model ready + CONTINUOUS BATCHING: max_batch=32, window=0ms (PMAT-044) + +CUDA-optimized server listening on http://127.0.0.1:19991 + +Endpoints: + GET /health - Health check + GET /metrics - Prometheus metrics + POST /generate - Text generation + POST /v1/completions - OpenAI-compatible + POST /v1/chat/completions - Chat completions + +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/started.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/started.txt new file mode 100644 index 0000000000..15b5e4b0e1 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/started.txt @@ -0,0 +1 @@ +2026-09-21T18:18:17Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/stderr.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/stderr.txt new file mode 100644 index 0000000000..e976d0bf04 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/stderr.txt @@ -0,0 +1,3 @@ +Launched apr serve on port 19991 (pid 1683636) +apr serve ready (15.5s) +Error: driver error: network error: apr serve HTTP 500: {"error":"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path"} diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/stdout.json b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/stdout.json new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/unittest.txt b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/unittest.txt new file mode 100644 index 0000000000..4434172cd8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/gx10-9B/unittest.txt @@ -0,0 +1,13 @@ +F. +====================================================================== +FAIL: test_mean (test_stats.TestStats.test_mean) +---------------------------------------------------------------------- +Traceback (most recent call last): + File "/home/noah/wt-3719-logs/cells/gx10-9B/project/test_stats.py", line 8, in test_mean + self.assertEqual(mean([2, 4, 6]), 4) +AssertionError: 6.0 != 4 + +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +FAILED (failures=1) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/apr-version.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/apr-version.txt new file mode 100644 index 0000000000..f571b481b8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (856009cc9) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/cell.json b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/cell.json new file mode 100644 index 0000000000..b14834c9ff --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/cell.json @@ -0,0 +1,27 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.0 (856009cc9)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-21T17:38:45Z", + "finished": "2026-09-21T17:45:39Z", + "rc": 1, + "test_rc": 1, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "unknown", + "fallback": false, + "evidence": "Model ready: 0 layers, vocab_size=248320, hidden_dim=2560\ngpu-layers: requested=all resolved=0 total=0 (backend=cuda)", + "model_layers": 0, + "answer_chars": 0, + "thinking": "unknown", + "thinking_raw": "", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [], + "verdict": "fail", + "mechanism": "serve child did not load", + "reason": "the child reported ready with 0 layers; first request: HTTP 500 {\"error\":\"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path\"}" +} diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/finished.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/finished.txt new file mode 100644 index 0000000000..506614aee8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/finished.txt @@ -0,0 +1 @@ +2026-09-21T17:45:39Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/hostname.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/lock-acquired b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/lock-acquired new file mode 100644 index 0000000000..d266a69720 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T17:45:24Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/model.sha256 b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stderr b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stderr new file mode 100644 index 0000000000..8777ad05b2 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stderr @@ -0,0 +1,6 @@ +[GH-129] Early kernel preload: 44 modules compiled +[PMAT-044] Batch scheduler started: max_batch=32, window=0ms +[PMAT-044] Batch m=1 done in 0.0ms (28800184.3 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (43487714.7 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (51977753.5 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (49897709.7 tok/s/slot) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stdout b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stdout new file mode 100644 index 0000000000..88ea21576f --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/serve-child.stdout @@ -0,0 +1,33 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:20033 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Building quantized inference model... +Model ready: 0 layers, vocab_size=248320, hidden_dim=2560 +gpu-layers: requested=all resolved=0 total=0 (backend=cuda) +Enabling optimized CUDA acceleration (PAR-111)... + Max sequence length: 4096 + Initializing GPU on device 0... + Pre-uploaded 497 MB weights to GPU +CUDA optimized model ready + CONTINUOUS BATCHING: max_batch=32, window=0ms (PMAT-044) + +CUDA-optimized server listening on http://127.0.0.1:20033 + +Endpoints: + GET /health - Health check + GET /metrics - Prometheus metrics + POST /generate - Text generation + POST /v1/completions - OpenAI-compatible + POST /v1/chat/completions - Chat completions + +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/started.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/started.txt new file mode 100644 index 0000000000..67d9b775af --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/started.txt @@ -0,0 +1 @@ +2026-09-21T17:38:45Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/stderr.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/stderr.txt new file mode 100644 index 0000000000..3f83bf3f78 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/stderr.txt @@ -0,0 +1,3 @@ +Launched apr serve on port 20033 (pid 3862672) +apr serve ready (7.0s) +Error: driver error: network error: apr serve HTTP 500: {"error":"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path"} diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/stdout.json b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/stdout.json new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/unittest.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/unittest.txt new file mode 100644 index 0000000000..bc6a505448 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-4B/unittest.txt @@ -0,0 +1,14 @@ +F. +====================================================================== +FAIL: test_mean (test_stats.TestStats.test_mean) +---------------------------------------------------------------------- +Traceback (most recent call last): + File "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells/lambda-4B/project/test_stats.py", line 8, in test_mean + self.assertEqual(mean([2, 4, 6]), 4) + ~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^ +AssertionError: 6.0 != 4 + +---------------------------------------------------------------------- +Ran 2 tests in 0.001s + +FAILED (failures=1) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/apr-version.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/apr-version.txt new file mode 100644 index 0000000000..f571b481b8 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (856009cc9) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/cell.json b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/cell.json new file mode 100644 index 0000000000..c6471359f5 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/cell.json @@ -0,0 +1,27 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-9B-Q4_K_M.gguf", + "sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", + "apr_version": "apr 0.69.0 (856009cc9)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-21T17:45:52Z", + "finished": "2026-09-21T17:59:08Z", + "rc": 1, + "test_rc": 1, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "unknown", + "fallback": false, + "evidence": "Model ready: 0 layers, vocab_size=248320, hidden_dim=4096\ngpu-layers: requested=all resolved=0 total=0 (backend=cuda)", + "model_layers": 0, + "answer_chars": 0, + "thinking": "unknown", + "thinking_raw": "", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [], + "verdict": "fail", + "mechanism": "serve child did not load", + "reason": "the child reported ready with 0 layers; first request: HTTP 500 {\"error\":\"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path\"}" +} diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/finished.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/finished.txt new file mode 100644 index 0000000000..001b9873d6 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/finished.txt @@ -0,0 +1 @@ +2026-09-21T17:59:08Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/hostname.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/lock-acquired b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/lock-acquired new file mode 100644 index 0000000000..d879c76f91 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T17:58:54Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/model.sha256 b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/model.sha256 new file mode 100644 index 0000000000..a25e916189 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/model.sha256 @@ -0,0 +1 @@ +03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8 diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stderr b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stderr new file mode 100644 index 0000000000..d4b1dc044d --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stderr @@ -0,0 +1,6 @@ +[GH-129] Early kernel preload: 44 modules compiled +[PMAT-044] Batch scheduler started: max_batch=32, window=0ms +[PMAT-044] Batch m=1 done in 0.0ms (37257824.1 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (56126171.6 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (54089138.9 tok/s/slot) +[PMAT-044] Batch m=1 done in 0.0ms (56827868.4 tok/s/slot) diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stdout b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stdout new file mode 100644 index 0000000000..3834d7e343 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/serve-child.stdout @@ -0,0 +1,33 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-9B-Q4_K_M.gguf +Binding: 127.0.0.1:19465 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 427 tensors, 46 metadata entries +Building quantized inference model... +Model ready: 0 layers, vocab_size=248320, hidden_dim=4096 +gpu-layers: requested=all resolved=0 total=0 (backend=cuda) +Enabling optimized CUDA acceleration (PAR-111)... + Max sequence length: 4096 + Initializing GPU on device 0... + Pre-uploaded 795 MB weights to GPU +CUDA optimized model ready + CONTINUOUS BATCHING: max_batch=32, window=0ms (PMAT-044) + +CUDA-optimized server listening on http://127.0.0.1:19465 + +Endpoints: + GET /health - Health check + GET /metrics - Prometheus metrics + POST /generate - Text generation + POST /v1/completions - OpenAI-compatible + POST /v1/chat/completions - Chat completions + +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/started.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/started.txt new file mode 100644 index 0000000000..bae5579051 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/started.txt @@ -0,0 +1 @@ +2026-09-21T17:45:52Z diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/stderr.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/stderr.txt new file mode 100644 index 0000000000..3b3b0a7cb9 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/stderr.txt @@ -0,0 +1,3 @@ +Launched apr serve on port 19465 (pid 1005393) +apr serve ready (6.5s) +Error: driver error: network error: apr serve HTTP 500: {"error":"Operation 'generate_gpu_resident_streaming' not supported: Model architecture not supported for GPU-resident path"} diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/stdout.json b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/stdout.json new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/unittest.txt b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/unittest.txt new file mode 100644 index 0000000000..19a5097517 --- /dev/null +++ b/evidence/apr-code-3719/baseline-856009cc9/lambda-9B/unittest.txt @@ -0,0 +1,14 @@ +F. +====================================================================== +FAIL: test_mean (test_stats.TestStats.test_mean) +---------------------------------------------------------------------- +Traceback (most recent call last): + File "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells/lambda-9B/project/test_stats.py", line 8, in test_mean + self.assertEqual(mean([2, 4, 6]), 4) + ~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^ +AssertionError: 6.0 != 4 + +---------------------------------------------------------------------- +Ran 2 tests in 0.001s + +FAILED (failures=1) diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/apr-version.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/apr-version.txt new file mode 100644 index 0000000000..86436ae9be --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (cc3892acd) diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/cell.json b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/cell.json new file mode 100644 index 0000000000..d953a1941b --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/cell.json @@ -0,0 +1,30 @@ +{ + "verb": "code", + "host": "gx10", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.0 (cc3892acd)", + "hostname": "gx10-a5b5", + "started": "2026-09-21T18:06:25Z", + "finished": "2026-09-21T18:19:01Z", + "rc": 0, + "test_rc": 1, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "The bug in `stats.py` is likely in the `test_mean` function where the mean is calculated incorrectly, possibly by not summing the values or dividing by the wrong count. Without seeing the actual code, I'll assume a common issue: the mean is calculated as `sum(values) / len(values)` but the function ", + "model_layers": 32, + "answer_chars": 1208, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [], + "files_added": [], + "files_removed": [], + "edit": [], + "verdict": "fail", + "mechanism": "wrong edit", + "reason": "stats.py was not changed and no unparsed call was left" +} diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/fake-serve-request-capture.jsonl b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/fake-serve-request-capture.jsonl new file mode 100644 index 0000000000..e22ac245fe --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/fake-serve-request-capture.jsonl @@ -0,0 +1 @@ +{"path": "/v1/chat/completions", "body": {"model": "Qwen3.5-4B-Q4_K_M", "messages": [{"role": "system", "content": "Answer the question. Be direct."}, {"role": "user", "content": "The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."}], "max_tokens": 1024, "temperature": 0.0, "stream": false}} diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/finished.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/finished.txt new file mode 100644 index 0000000000..671c731c88 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/finished.txt @@ -0,0 +1 @@ +2026-09-21T18:19:01Z diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/hostname.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/hostname.txt new file mode 100644 index 0000000000..c9c46b28f5 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/hostname.txt @@ -0,0 +1 @@ +gx10-a5b5 diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/lock-acquired b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/lock-acquired new file mode 100644 index 0000000000..189078c2b4 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T18:18:41Z diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/model.sha256 b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stderr b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stderr new file mode 100644 index 0000000000..dcf0a7430e --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stderr @@ -0,0 +1,25 @@ +Backend: GPU (CUDA, NVIDIA GB10, 122502 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: receipt matches (model sha256 00fe7986ff5f…, apr 0.69.0, NVIDIA GB10) — validated 11656s ago on 16 positions; CPU reference forward skipped [source=receipt, sha256 4788 ms]. `apr run --revalidate` forces a fresh run. +[GH-559] RmsNorm PTX (1888 bytes) +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 5 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stdout b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stdout new file mode 100644 index 0000000000..084038b34c --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/serve-child.stdout @@ -0,0 +1,63 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:19483 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:19483 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + POST /v1/batch/completions + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/started.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/started.txt new file mode 100644 index 0000000000..a1cca97cd4 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/started.txt @@ -0,0 +1 @@ +2026-09-21T18:06:25Z diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stats.diff b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stats.diff new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stderr.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stderr.txt new file mode 100644 index 0000000000..fc2206e4d5 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stderr.txt @@ -0,0 +1,2 @@ +Launched apr serve on port 19483 (pid 704102) +apr serve ready (8.0s) diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stdout.json b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stdout.json new file mode 100644 index 0000000000..3cf2d5b62c --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"duration_ms":11768,"result":"The bug in `stats.py` is likely in the `test_mean` function where the mean is calculated incorrectly, possibly by not summing the values or dividing by the wrong count. Without seeing the actual code, I'll assume a common issue: the mean is calculated as `sum(values) / len(values)` but the function might be using `sum(values) / 1` or similar.\n\nHowever, since I don't have the actual code, I'll provide a generic fix based on common issues:\n\n```python\n# In stats.py, find the mean calculation and ensure it's:\n# mean = sum(values) / len(values)\n```\n\nAfter fixing, run the tests:\n\n```bash\npython3 -m unittest test_stats -v\n```\n\nIf all tests pass, report that they do. If not, provide the error message.\n\n**Note:** Since I don't have the actual code, I cannot provide the exact one-line edit. Please provide the code for `stats.py` and `test_stats.py` for a precise fix.\n\n**Assuming the common issue:**\n\nIf the `mean` function is defined as:\n```python\ndef mean(values):\n return sum(values) / 1 # Bug: dividing by 1 instead of len(values)\n```\n\nThe fix would be:\n```python\ndef mean(values):\n return sum(values) / len(values)\n```\n\nThis is a one-line edit. After applying this fix, run the tests to verify.","session_id":"00000000-0006-7000-5c02-5c0248cafed5","num_turns":1,"tokens_in":81,"tokens_out":314,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/trace.jsonl b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/trace.jsonl new file mode 100644 index 0000000000..886793b695 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c02-5c0248caff23","ts":"@1790014741","actor":"apr-code","model":"/home/noah/models/Qwen3.5-4B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"The bug in `stats.py` is likely in the `test_mean` function where the mean is calculated incorrectly, possibly by not summing the values or dividing by the wrong count. Without seeing the actual code, I'll assume a common issue: the mean is calculated as `sum(values) / len(values)` but the function might be using `sum(values) / 1` or similar.\n\nHowever, since I don't have the actual code, I'll provide a generic fix based on common issues:\n\n```python\n# In stats.py, find the mean calculation and ensure it's:\n# mean = sum(values) / len(values)\n```\n\nAfter fixing, run the tests:\n\n```bash\npython3 -m unittest test_stats -v\n```\n\nIf all tests pass, report that they do. If not, provide the error message.\n\n**Note:** Since I don't have the actual code, I cannot provide the exact one-line edit. Please provide the code for `stats.py` and `test_stats.py` for a precise fix.\n\n**Assuming the common issue:**\n\nIf the `mean` function is defined as:\n```python\ndef mean(values):\n return sum(values) / 1 # Bug: dividing by 1 instead of len(values)\n```\n\nThe fix would be:\n```python\ndef mean(values):\n return sum(values) / len(values)\n```\n\nThis is a one-line edit. After applying this fix, run the tests to verify."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":11768,"tokens_in":81,"tokens_out":314} diff --git a/evidence/apr-code-3719/route-cc3892acd/gx10-4B/unittest.txt b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/unittest.txt new file mode 100644 index 0000000000..83614f2251 --- /dev/null +++ b/evidence/apr-code-3719/route-cc3892acd/gx10-4B/unittest.txt @@ -0,0 +1,13 @@ +F. +====================================================================== +FAIL: test_mean (test_stats.TestStats.test_mean) +---------------------------------------------------------------------- +Traceback (most recent call last): + File "/home/noah/wt-3719-logs/cells-v2/gx10-4B/project/test_stats.py", line 8, in test_mean + self.assertEqual(mean([2, 4, 6]), 4) +AssertionError: 6.0 != 4 + +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +FAILED (failures=1) diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/apr-version.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/apr-version.txt new file mode 100644 index 0000000000..a7da1a8ca7 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (7b161d743) diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/cell.json b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/cell.json new file mode 100644 index 0000000000..2650234669 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "gx10", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.0 (7b161d743)", + "hostname": "gx10-a5b5", + "started": "2026-09-21T21:49:58Z", + "finished": "2026-09-21T22:06:08Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 299, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/home/noah/wt-3719-logs/cells-v6/gx10-4B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/home/noah/wt-3719-logs/cells-v6/gx10-4B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/finished.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/finished.txt new file mode 100644 index 0000000000..a05097bb95 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/finished.txt @@ -0,0 +1 @@ +2026-09-21T22:06:08Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/hostname.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/hostname.txt new file mode 100644 index 0000000000..c9c46b28f5 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/hostname.txt @@ -0,0 +1 @@ +gx10-a5b5 diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/lock-acquired b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/lock-acquired new file mode 100644 index 0000000000..9554ac1264 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T22:03:22Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/model.sha256 b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stderr b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stderr new file mode 100644 index 0000000000..83fa54af8d --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stderr @@ -0,0 +1,25 @@ +Backend: GPU (CUDA, NVIDIA GB10, 122502 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: receipt matches (model sha256 00fe7986ff5f…, apr 0.69.0, NVIDIA GB10) — validated 25139s ago on 16 positions; CPU reference forward skipped [source=receipt, sha256 4783 ms]. `apr run --revalidate` forces a fresh run. +[GH-559] RmsNorm PTX (1888 bytes) +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 5 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stdout b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stdout new file mode 100644 index 0000000000..5bc30e6158 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/serve-child.stdout @@ -0,0 +1,63 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:20259 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:20259 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + POST /v1/batch/completions + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/started.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/started.txt new file mode 100644 index 0000000000..e5c22d1d7e --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/started.txt @@ -0,0 +1 @@ +2026-09-21T21:49:58Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stats.diff b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stderr.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stderr.txt new file mode 100644 index 0000000000..086f08ff5c --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stderr.txt @@ -0,0 +1,16 @@ +gpu-q: waiting (prio 1, 6/8 in queue) +gpu-q: waiting (prio 1, 6/9 in queue) +gpu-q: waiting (prio 1, 6/9 in queue) +gpu-q: waiting (prio 1, 5/8 in queue) +gpu-q: waiting (prio 1, 4/8 in queue) +gpu-q: waiting (prio 1, 4/8 in queue) +gpu-q: waiting (prio 1, 3/7 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +gpu-q: waiting (prio 1, 2/8 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +gpu-q: waiting (prio 1, 2/10 in queue) +Launched apr serve on port 20259 (pid 3659888) +apr serve ready (9.5s) diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stdout.json b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stdout.json new file mode 100644 index 0000000000..767a7e3dda --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"duration_ms":155275,"result":"The bug was in the `mean` function in `stats.py`. It was dividing by `len(values) - 1` instead of `len(values)`, which caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\nThe fix was to change `len(values) - 1` to `len(values)`.\n\nAll tests now pass.","session_id":"00000000-0006-7000-5c05-5c0574fe462c","num_turns":5,"tokens_in":4796,"tokens_out":220,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/trace.jsonl b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/trace.jsonl new file mode 100644 index 0000000000..2e196be38b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c05-5c0574fe4661","ts":"@1790028367","actor":"apr-code","model":"/home/noah/models/Qwen3.5-4B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"The bug was in the `mean` function in `stats.py`. It was dividing by `len(values) - 1` instead of `len(values)`, which caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\nThe fix was to change `len(values) - 1` to `len(values)`.\n\nAll tests now pass."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":155275,"tokens_in":4796,"tokens_out":220} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-4B/unittest.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/unittest.txt new file mode 100644 index 0000000000..c81d30b11f --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-4B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +OK diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/apr-version.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/apr-version.txt new file mode 100644 index 0000000000..a7da1a8ca7 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (7b161d743) diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/cell.json b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/cell.json new file mode 100644 index 0000000000..1dde6455bd --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "gx10", + "file": "Qwen3.5-9B-Q4_K_M.gguf", + "sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", + "apr_version": "apr 0.69.0 (7b161d743)", + "hostname": "gx10-a5b5", + "started": "2026-09-21T22:06:11Z", + "finished": "2026-09-21T22:22:29Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 261, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/home/noah/wt-3719-logs/cells-v6/gx10-9B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/home/noah/wt-3719-logs/cells-v6/gx10-9B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/finished.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/finished.txt new file mode 100644 index 0000000000..4feef3aed4 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/finished.txt @@ -0,0 +1 @@ +2026-09-21T22:22:29Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/hostname.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/hostname.txt new file mode 100644 index 0000000000..c9c46b28f5 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/hostname.txt @@ -0,0 +1 @@ +gx10-a5b5 diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/lock-acquired b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/lock-acquired new file mode 100644 index 0000000000..9219828529 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T22:18:30Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/model.sha256 b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/model.sha256 new file mode 100644 index 0000000000..a25e916189 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/model.sha256 @@ -0,0 +1 @@ +03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8 diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stderr b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stderr new file mode 100644 index 0000000000..96dd795c66 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stderr @@ -0,0 +1,24 @@ +Backend: GPU (CUDA, NVIDIA GB10, 122502 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: receipt matches (model sha256 03b74727a860…, apr 0.69.0, NVIDIA GB10) — validated 7062s ago on 65 positions; CPU reference forward skipped [source=receipt, sha256 10244 ms]. `apr run --revalidate` forces a fresh run. +[GH-559] RmsNorm PTX (1888 bytes) +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 2 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 5 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround +[GH-480] Patched 1 backward branch(es) for sm_121 JIT workaround diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stdout b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stdout new file mode 100644 index 0000000000..a79846bf01 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/serve-child.stdout @@ -0,0 +1,63 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-9B-Q4_K_M.gguf +Binding: 127.0.0.1:20110 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 427 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:20110 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + POST /v1/batch/completions + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/started.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/started.txt new file mode 100644 index 0000000000..485fbb0928 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/started.txt @@ -0,0 +1 @@ +2026-09-21T22:06:11Z diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stats.diff b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stderr.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stderr.txt new file mode 100644 index 0000000000..c1f8b051d3 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stderr.txt @@ -0,0 +1,15 @@ +gpu-q: waiting (prio 1, 7/10 in queue) +gpu-q: waiting (prio 1, 7/10 in queue) +gpu-q: waiting (prio 1, 6/9 in queue) +gpu-q: waiting (prio 1, 6/9 in queue) +gpu-q: waiting (prio 1, 5/8 in queue) +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 3/6 in queue) +gpu-q: waiting (prio 1, 3/6 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +gpu-q: waiting (prio 1, 2/5 in queue) +Launched apr serve on port 20110 (pid 3730730) +apr serve ready (23.5s) diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stdout.json b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stdout.json new file mode 100644 index 0000000000..12e7e5eab9 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"duration_ms":214191,"result":"All tests pass. The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is an off-by-one error that would give incorrect results for the arithmetic mean. The fix changes it to divide by `len(values)`.","session_id":"00000000-0006-7000-5c05-5c05af721e77","num_turns":4,"tokens_in":3970,"tokens_out":191,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/trace.jsonl b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/trace.jsonl new file mode 100644 index 0000000000..28986edcf3 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c05-5c05af721ec1","ts":"@1790029348","actor":"apr-code","model":"/home/noah/models/Qwen3.5-9B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"All tests pass. The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is an off-by-one error that would give incorrect results for the arithmetic mean. The fix changes it to divide by `len(values)`."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":214191,"tokens_in":3970,"tokens_out":191} diff --git a/evidence/apr-code-3719/v6-7b161d743/gx10-9B/unittest.txt b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/unittest.txt new file mode 100644 index 0000000000..c81d30b11f --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/gx10-9B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +OK diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/apr-version.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/apr-version.txt new file mode 100644 index 0000000000..a7da1a8ca7 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (7b161d743) diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/cell.json b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/cell.json new file mode 100644 index 0000000000..a4cc16d7d6 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.0 (7b161d743)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-21T21:50:15Z", + "finished": "2026-09-21T21:55:34Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 285, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells-v6/lambda-4B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells-v6/lambda-4B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/finished.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/finished.txt new file mode 100644 index 0000000000..c3159733a0 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/finished.txt @@ -0,0 +1 @@ +2026-09-21T21:55:34Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/hostname.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/lock-acquired b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/lock-acquired new file mode 100644 index 0000000000..1c9a7f49a3 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T21:54:01Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/model.sha256 b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stderr b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stderr new file mode 100644 index 0000000000..eabf1eb7d4 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stderr @@ -0,0 +1,2 @@ +Backend: GPU (CUDA, NVIDIA GeForce RTX 4090, 24035 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: receipt matches (model sha256 00fe7986ff5f…, apr 0.69.0, NVIDIA GeForce RTX 4090) — validated 24780s ago on 16 positions; CPU reference forward skipped [source=receipt, sha256 1185 ms]. `apr run --revalidate` forces a fresh run. diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stdout b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stdout new file mode 100644 index 0000000000..0c7fcf1c19 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/serve-child.stdout @@ -0,0 +1,63 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:19978 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:19978 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + POST /v1/batch/completions + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/started.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/started.txt new file mode 100644 index 0000000000..efa423add9 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/started.txt @@ -0,0 +1 @@ +2026-09-21T21:50:15Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stats.diff b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stderr.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stderr.txt new file mode 100644 index 0000000000..bdfddd7d6a --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stderr.txt @@ -0,0 +1,6 @@ +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 4/7 in queue) +gpu-q: waiting (prio 1, 3/6 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +Launched apr serve on port 19978 (pid 228598) +apr serve ready (10.5s) diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stdout.json b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stdout.json new file mode 100644 index 0000000000..fbd4fd3bfc --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"duration_ms":80571,"result":"The bug was in the `mean` function. It was dividing by `len(values) - 1` instead of `len(values)`, which caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\nThe fix was to change `len(values) - 1` to `len(values)`.\n\nAll tests now pass.","session_id":"00000000-0006-7000-5c05-5c054f26c726","num_turns":5,"tokens_in":6524,"tokens_out":216,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/trace.jsonl b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/trace.jsonl new file mode 100644 index 0000000000..5d07800ef5 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c05-5c054f26c7b5","ts":"@1790027732","actor":"apr-code","model":"/home/noah/models/Qwen3.5-4B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"The bug was in the `mean` function. It was dividing by `len(values) - 1` instead of `len(values)`, which caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\nThe fix was to change `len(values) - 1` to `len(values)`.\n\nAll tests now pass."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":80571,"tokens_in":6524,"tokens_out":216} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-4B/unittest.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/unittest.txt new file mode 100644 index 0000000000..84ccb04487 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-4B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.005s + +OK diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/apr-version.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/apr-version.txt new file mode 100644 index 0000000000..a7da1a8ca7 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.0 (7b161d743) diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/cell.json b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/cell.json new file mode 100644 index 0000000000..3c0e055d30 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-9B-Q4_K_M.gguf", + "sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", + "apr_version": "apr 0.69.0 (7b161d743)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-21T21:55:57Z", + "finished": "2026-09-21T22:04:12Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 326, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells-v6/lambda-9B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/tmp/claude-1000/-home-noah-src-aprender/f037f0cc-1719-460c-88ec-74b25b6f89a9/scratchpad/3719/cells-v6/lambda-9B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/finished.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/finished.txt new file mode 100644 index 0000000000..a1c5610486 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/finished.txt @@ -0,0 +1 @@ +2026-09-21T22:04:12Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/hostname.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/lock-acquired b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/lock-acquired new file mode 100644 index 0000000000..4e63e72075 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/lock-acquired @@ -0,0 +1 @@ +2026-09-21T22:02:05Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/model.sha256 b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/model.sha256 new file mode 100644 index 0000000000..a25e916189 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/model.sha256 @@ -0,0 +1 @@ +03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8 diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stderr b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stderr new file mode 100644 index 0000000000..d8d5d6d738 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stderr @@ -0,0 +1,3 @@ +Backend: GPU (CUDA, NVIDIA GeForce RTX 4090, 24035 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: validating on this run (receipt schema 2, this build writes 1) [source=fresh] +F2 guard: passed in 15020 ms on 65 positions; receipt written to /home/noah/.cache/apr/f2-receipts/03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8.json — the next run of this (model, apr, device) skips it. diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stdout b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stdout new file mode 100644 index 0000000000..fb01e2b7ba --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/serve-child.stdout @@ -0,0 +1,63 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-9B-Q4_K_M.gguf +Binding: 127.0.0.1:19722 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 427 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:19722 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + POST /v1/batch/completions + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/started.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/started.txt new file mode 100644 index 0000000000..bfede63ccc --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/started.txt @@ -0,0 +1 @@ +2026-09-21T21:55:57Z diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stats.diff b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stderr.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stderr.txt new file mode 100644 index 0000000000..91043a9242 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stderr.txt @@ -0,0 +1,9 @@ +gpu-q: waiting (prio 1, 3/6 in queue) +gpu-q: waiting (prio 1, 3/6 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +gpu-q: waiting (prio 1, 2/6 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +gpu-q: waiting (prio 1, 2/9 in queue) +Launched apr serve on port 19722 (pid 770417) +apr serve ready (11.0s) diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stdout.json b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stdout.json new file mode 100644 index 0000000000..e0fe1db7b5 --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"duration_ms":114491,"result":"All tests pass. The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is a common mistake confusing the arithmetic mean with the sample mean (which uses n-1 for variance calculations). The fix changes it to divide by `len(values)` for the correct arithmetic mean.","session_id":"00000000-0006-7000-5c05-5c056e101295","num_turns":4,"tokens_in":5370,"tokens_out":196,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/trace.jsonl b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/trace.jsonl new file mode 100644 index 0000000000..da58e80c9b --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c05-5c056e1012ce","ts":"@1790028251","actor":"apr-code","model":"/home/noah/models/Qwen3.5-9B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"All tests pass. The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is a common mistake confusing the arithmetic mean with the sample mean (which uses n-1 for variance calculations). The fix changes it to divide by `len(values)` for the correct arithmetic mean."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":114491,"tokens_in":5370,"tokens_out":196} diff --git a/evidence/apr-code-3719/v6-7b161d743/lambda-9B/unittest.txt b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/unittest.txt new file mode 100644 index 0000000000..c81d30b11f --- /dev/null +++ b/evidence/apr-code-3719/v6-7b161d743/lambda-9B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +OK diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/apr-version.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/apr-version.txt new file mode 100644 index 0000000000..23691853fd --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.3 (bb79ae2c66) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/cell.json b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/cell.json new file mode 100644 index 0000000000..b7bc0ea8ed --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-4B-Q4_K_M.gguf", + "sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", + "apr_version": "apr 0.69.3 (bb79ae2c66)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-27T09:58:09Z", + "finished": "2026-09-27T09:58:33Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 332, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/mnt/nvme-raid0/aprender-wt/c1-3719/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/mnt/nvme-raid0/aprender-wt/c1-3719/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/finished.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/finished.txt new file mode 100644 index 0000000000..3b0b5cf6d4 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/finished.txt @@ -0,0 +1 @@ +2026-09-27T09:58:33Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/hostname.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/lock-acquired b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/lock-acquired new file mode 100644 index 0000000000..06a4cf503b --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/lock-acquired @@ -0,0 +1 @@ +2026-09-27T09:58:09Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/model.sha256 b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/model.sha256 new file mode 100644 index 0000000000..43fc99f2ba --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/model.sha256 @@ -0,0 +1 @@ +00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4 diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stderr b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stderr new file mode 100644 index 0000000000..5a22803c1e --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stderr @@ -0,0 +1,8 @@ +Backend: GPU (CUDA, NVIDIA GeForce RTX 4090, 24035 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: validating on this run (receipt written by apr 0.70.0 exe:088bf48f4aa8c317, this is 0.69.3) [source=fresh] +F2 guard: passed in 12400 ms on 65 positions; receipt written to /home/noah/.cache/apr/f2-receipts/00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4.json — the next run of this (model, apr, device) skips it. +[qwen35] batched prefill: 980 tokens in 439 ms (2233 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1150 tokens in 455 ms (2530 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1325 tokens in 456 ms (2906 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1456 tokens in 465 ms (3132 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1545 tokens in 541 ms (2857 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stdout b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stdout new file mode 100644 index 0000000000..4122f63d37 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/serve-child.stdout @@ -0,0 +1,62 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-4B-Q4_K_M.gguf +Binding: 127.0.0.1:44579 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 426 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:44579 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/started.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/started.txt new file mode 100644 index 0000000000..06a4cf503b --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/started.txt @@ -0,0 +1 @@ +2026-09-27T09:58:09Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stats.diff b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stderr.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stderr.txt new file mode 100644 index 0000000000..fde8d4b681 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stderr.txt @@ -0,0 +1,2 @@ +Launched apr serve on port 44579 (pid 403832, --gpu) +apr serve ready (5.0s) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stdout.json b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stdout.json new file mode 100644 index 0000000000..84ef940cd8 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"status":"ok","duration_ms":19025,"result":"All tests pass.\n\n**Bug found:** In `stats.py`, the `mean()` function was dividing by `len(values) - 1` instead of `len(values)`. This caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\n**Fix:** Changed `return sum(values) / (len(values) - 1)` to `return sum(values) / len(values)`.","session_id":"00000000-0006-7000-5c73-5c73fe08cf52","num_turns":5,"tokens_in":6456,"tokens_out":230,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/trace.jsonl b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/trace.jsonl new file mode 100644 index 0000000000..2386451535 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c73-5c73fe08cf8b","ts":"@1790503113","actor":"apr-code","model":"/home/noah/models/Qwen3.5-4B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"All tests pass.\n\n**Bug found:** In `stats.py`, the `mean()` function was dividing by `len(values) - 1` instead of `len(values)`. This caused the mean of `[2, 4, 6]` to be calculated as `12 / 2 = 6` instead of the correct `12 / 3 = 4`.\n\n**Fix:** Changed `return sum(values) / (len(values) - 1)` to `return sum(values) / len(values)`."}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":19025,"tokens_in":6456,"tokens_out":230} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/unittest.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/unittest.txt new file mode 100644 index 0000000000..c81d30b11f --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-4B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +OK diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/apr-version.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/apr-version.txt new file mode 100644 index 0000000000..23691853fd --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/apr-version.txt @@ -0,0 +1 @@ +apr 0.69.3 (bb79ae2c66) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/cell.json b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/cell.json new file mode 100644 index 0000000000..1b0ee1b38d --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/cell.json @@ -0,0 +1,36 @@ +{ + "verb": "code", + "host": "lambda", + "file": "Qwen3.5-9B-Q4_K_M.gguf", + "sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", + "apr_version": "apr 0.69.3 (bb79ae2c66)", + "hostname": "noah-Lambda-Vector", + "started": "2026-09-27T09:59:08Z", + "finished": "2026-09-27T09:59:48Z", + "rc": 0, + "test_rc": 0, + "task": "tests/fixtures/apr-code-edit-verify", + "backend": "cuda", + "fallback": false, + "evidence": "Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens", + "model_layers": 32, + "answer_chars": 416, + "thinking": "off", + "thinking_raw": "off", + "prompt_tokens": null, + "context": "task", + "max_tokens": 1024, + "agent_python_calls": [ + "/mnt/nvme-raid0/aprender-wt/c1-3719/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/project\t-m unittest test_stats -v" + ], + "files_added": [], + "files_removed": [], + "edit": [ + "- return sum(values) / (len(values) - 1)", + "+ return sum(values) / len(values)" + ], + "test_call": "/mnt/nvme-raid0/aprender-wt/c1-3719/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/project\t-m unittest test_stats -v", + "verdict": "pass", + "mechanism": "", + "reason": "" +} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/finished.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/finished.txt new file mode 100644 index 0000000000..9fc5d7aa37 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/finished.txt @@ -0,0 +1 @@ +2026-09-27T09:59:48Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/hostname.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/hostname.txt new file mode 100644 index 0000000000..3c78b077cf --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/hostname.txt @@ -0,0 +1 @@ +noah-Lambda-Vector diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/lock-acquired b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/lock-acquired new file mode 100644 index 0000000000..f8533d87fe --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/lock-acquired @@ -0,0 +1 @@ +2026-09-27T09:59:08Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/model.sha256 b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/model.sha256 new file mode 100644 index 0000000000..a25e916189 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/model.sha256 @@ -0,0 +1 @@ +03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8 diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stderr b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stderr new file mode 100644 index 0000000000..02b93030e0 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stderr @@ -0,0 +1,7 @@ +Backend: GPU (CUDA, NVIDIA GeForce RTX 4090, 24035 MB VRAM) [qwen35 hybrid forward, #3090] +F2 guard: validating on this run (receipt written by apr 0.70.0 exe:c61f63c16803814d, this is 0.69.3) [source=fresh] +F2 guard: passed in 15464 ms on 65 positions; receipt written to /home/noah/.cache/apr/f2-receipts/03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8.json — the next run of this (model, apr, device) skips it. +[qwen35] batched prefill: 980 tokens in 501 ms (1957 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1325 tokens in 671 ms (1976 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1456 tokens in 723 ms (2015 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) +[qwen35] batched prefill: 1545 tokens in 888 ms (1741 tok/s, chunk 512 rows, attention cuBLAS f32, from position 0) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stdout b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stdout new file mode 100644 index 0000000000..f29f9813dd --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/serve-child.stdout @@ -0,0 +1,62 @@ +=== APR Serve === + +Model: /home/noah/models/Qwen3.5-9B-Q4_K_M.gguf +Binding: 127.0.0.1:36865 + +Model loading: mmap + +Press Ctrl+C to stop + +Detected format: GGUF +Starting GGUF inference server... +Loading GGUF model (mmap)... +GGUF loaded: 427 tensors, 46 metadata entries +Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens +chat template: Qwen3NoThink (thinking off) +gpu-layers: requested=all resolved=32 total=32 (backend=cuda) + +Inference server listening on http://127.0.0.1:36865 + +Endpoints: + GET / + GET /health + GET /health/live + GET /health/ready + GET /ready + GET /models + POST /tokenize + POST /generate + POST /batch/tokenize + POST /batch/generate + POST /stream/generate + POST /realize/generate + POST /realize/batch + POST /realize/embed + GET /realize/model + POST /realize/reload + GET /metrics + GET /metrics/dispatch + POST /metrics/dispatch/reset + GET /v1/models + POST /v1/completions + POST /v1/chat/completions + POST /v1/chat/completions/stream + POST /v1/embeddings + POST /v1/predict + POST /v1/explain + GET /v1/audit/:request_id + POST /v1/gpu/warmup + GET /v1/gpu/status + GET /v1/metrics + GET /v1/effective-config + POST /api/chat + POST /api/generate + GET /api/tags + POST /api/show + GET /api/version + POST /api/embeddings + POST /v1/logprobs + POST /v1/perplexity + +Performance targets: 100+ tok/s CPU, 500+ tok/s GPU +Press Ctrl+C to stop diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/started.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/started.txt new file mode 100644 index 0000000000..f8533d87fe --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/started.txt @@ -0,0 +1 @@ +2026-09-27T09:59:08Z diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stats.diff b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stats.diff new file mode 100644 index 0000000000..e51e887c4b --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stats.diff @@ -0,0 +1,11 @@ +--- stats.py ++++ stats.py +@@ -3,7 +3,7 @@ + + def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" +- return sum(values) / (len(values) - 1) ++ return sum(values) / len(values) + + + def median(values): diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stderr.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stderr.txt new file mode 100644 index 0000000000..f2938073e3 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stderr.txt @@ -0,0 +1,2 @@ +Launched apr serve on port 36865 (pid 481090, --gpu) +apr serve ready (14.0s) diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stdout.json b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stdout.json new file mode 100644 index 0000000000..3ca999ae47 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/stdout.json @@ -0,0 +1 @@ +{"type":"result","subtype":"success","is_error":false,"status":"ok","duration_ms":23849,"result":"The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is an off-by-one error that incorrectly computes the mean as a biased estimator.\n\nThe fix changed `len(values) - 1` to `len(values)`, which is the correct formula for arithmetic mean.\n\nAll tests now pass:\n- `test_mean`: `mean([2, 4, 6])` correctly returns `4`\n- `test_median`: Both median tests pass","session_id":"00000000-0006-7000-5c74-5c7402692096","num_turns":4,"tokens_in":5306,"tokens_out":236,"total_cost_usd":0} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/trace.jsonl b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/trace.jsonl new file mode 100644 index 0000000000..194acef736 --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/trace.jsonl @@ -0,0 +1,4 @@ +{"v":1,"kind":"session_start","session_id":"00000000-0006-7000-5c74-5c74026920bf","ts":"@1790503186","actor":"apr-code","model":"/home/noah/models/Qwen3.5-9B-Q4_K_M.gguf","cwd_sha256":"0000000000000000000000000000000000000000000000000000000000000000"} +{"v":1,"kind":"user_prompt","turn":0,"text":"The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass."} +{"v":1,"kind":"assistant_turn","turn":1,"blocks":[{"type":"text","text":"The bug was in `stats.py` line 6: the mean function was dividing by `len(values) - 1` instead of `len(values)`. This is an off-by-one error that incorrectly computes the mean as a biased estimator.\n\nThe fix changed `len(values) - 1` to `len(values)`, which is the correct formula for arithmetic mean.\n\nAll tests now pass:\n- `test_mean`: `mean([2, 4, 6])` correctly returns `4`\n- `test_median`: Both median tests pass"}],"stop_reason":"end_turn"} +{"v":1,"kind":"session_end","turn":1,"stop_reason":"end_turn","elapsed_ms":23849,"tokens_in":5306,"tokens_out":236} diff --git a/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/unittest.txt b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/unittest.txt new file mode 100644 index 0000000000..c81d30b11f --- /dev/null +++ b/evidence/apr-code-3719/v7-bb79ae2c6/lambda-9B/unittest.txt @@ -0,0 +1,5 @@ +.. +---------------------------------------------------------------------- +Ran 2 tests in 0.000s + +OK diff --git a/scripts/apr_code_edit_verify.sh b/scripts/apr_code_edit_verify.sh new file mode 100755 index 0000000000..f29f0db40b --- /dev/null +++ b/scripts/apr_code_edit_verify.sh @@ -0,0 +1,185 @@ +#!/usr/bin/env bash +# apr_code_edit_verify.sh - does `apr code` finish a scripted edit-and-verify +# task on a real model, on CUDA? (#3719) +# +# One run is one cell: one model on one host. The fixture under +# tests/fixtures/apr-code-edit-verify/ is a small Python project whose +# test_mean fails; the fix is a one-line edit to stats.py. task.txt asks the +# agent to make that fix and run the test. +# +# Nothing the agent SAYS counts as evidence: +# - the edit is judged by diffing the working copy against the fixture; +# - the test is re-run HERE, after the agent exits; +# - "the agent ran the test" is read from logging python3/python shims put +# first on the agent's PATH (its shell tool runs `sh -c` with the +# inherited environment). The --emit-trace file cannot answer it: it +# holds one text block per run and never a tool call +# (crates/aprender-orchestrate/src/agent/code.rs emit_ccpa_trace), and +# the PreToolUse/PostToolUse hooks are not wired into the loop; +# - the backend is read from the serve CHILD's own output. `apr code` has no +# GPU flag: its driver spawns an `apr serve` child with `--gpu` +# (crates/aprender-orchestrate/src/agent/driver/apr_serve.rs), and a +# requested flag proves nothing. The driver pipes the child's output and +# shows it only when startup fails, so APR_BIN points it at a wrapper that +# runs the pinned binary and tees the child's stdout and stderr to files. +# +# The judge is scripts/lib/apr_code_edit_verify.py; it writes /cell.json. +# +# Usage: +# scripts/apr_code_edit_verify.sh --model FILE --host NAME --out DIR +# [--max-turns N] [--lock-wait SEC] [--timeout SEC] [--gpu-q PRIO] +# DIR must not exist yet; the run writes every artifact into it. +# --gpu-q PRIO queues the run through `gpu-q --prio PRIO` (the cop's ordered +# front of the same lock; release-blocking rows run at 1) instead of a bare +# bounded flock. gpu-q itself takes the lock and choom, so the harness must +# not take it again: two flocks on one file deadlock. The wait is bounded by +# GPUQ_WAIT=LOCK_WAIT (gpu-q v3, rule rev 6: exits 75 without running the +# command), and the run keeps its own `timeout TIMEOUT`. A gpu-q whose +# `--caps` does not list `wait` has no bound, so the harness falls back to +# the bounded flock, as scripts/model_ladder.sh does (#3771). +# +# Exit: 0 PASS; 1 FAIL (cell.json names the first failing mechanism); +# 2 decline (GPU lock not acquired, missing input); 3 usage error. +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +FIXTURE="$ROOT/tests/fixtures/apr-code-edit-verify" +JUDGE="$ROOT/scripts/lib/apr_code_edit_verify.py" +# APR_GPU_LOCK exists for the case table, which must never queue on the fleet +# lock; gpu-q reads the same path from GPUQ_LOCK. +GPU_LOCK="${APR_GPU_LOCK:-/tmp/apr-gpu.lock}" + +MODEL="" +HOST="" +OUT="" +MAX_TURNS=12 +LOCK_WAIT=3600 +TIMEOUT=900 +GPU_Q_PRIO="" + +usage() { + sed -n '/^# Usage:/,/^# Exit:/p' "$0" | sed 's/^# \{0,1\}//' >&2 + exit 3 +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --model) MODEL="${2:-}"; shift 2 ;; + --host) HOST="${2:-}"; shift 2 ;; + --out) OUT="${2:-}"; shift 2 ;; + --max-turns) MAX_TURNS="${2:-}"; shift 2 ;; + --lock-wait) LOCK_WAIT="${2:-}"; shift 2 ;; + --timeout) TIMEOUT="${2:-}"; shift 2 ;; + --gpu-q) GPU_Q_PRIO="${2:-}"; shift 2 ;; + -h|--help) usage ;; + *) printf 'unknown argument: %s\n' "$1" >&2; usage ;; + esac +done + +if [ -z "$MODEL" ] || [ -z "$HOST" ] || [ -z "$OUT" ]; then + usage +fi +if [ ! -f "$MODEL" ]; then + printf 'DECLINE model file not found: %s\n' "$MODEL" >&2 + exit 2 +fi +if ! command -v python3 >/dev/null 2>&1; then + printf 'DECLINE python3 not found: the fixture test and the judge need it\n' >&2 + exit 2 +fi +if [ -n "$GPU_Q_PRIO" ] && ! command -v gpu-q >/dev/null 2>&1; then + printf 'DECLINE --gpu-q given but gpu-q is not on PATH\n' >&2 + exit 2 +fi + +# shellcheck source=scripts/apr_bin.sh +. "$ROOT/scripts/apr_bin.sh" || exit 2 + +if [ -e "$OUT" ]; then + printf 'usage: --out %s already exists; give a fresh directory\n' "$OUT" >&2 + exit 3 +fi +mkdir -p "$OUT" +OUT="$(cd "$OUT" && pwd)" +cp -R "$FIXTURE/project" "$OUT/project" + +# The wrapper runs the pinned binary for the serve child and records what the +# child prints. tee writes back to the original streams, so the driver sees +# exactly what it would have seen without the wrapper. +wrapper="$OUT/apr-serve-child" +{ + printf '#!/usr/bin/env bash\n' + printf 'exec %q "$@" > >(tee -a %q) 2> >(tee -a %q >&2)\n' \ + "$APR" "$OUT/serve-child.stdout" "$OUT/serve-child.stderr" +} > "$wrapper" +chmod +x "$wrapper" + +# The shims record every python invocation the agent's shell makes (cwd and +# argv), then run the real interpreter. The harness's own re-run below calls +# the real interpreter by path, so it never appears in the log. +real_python="$(command -v python3)" +shim_dir="$OUT/shim" +mkdir -p "$shim_dir" +for name in python3 python; do + { + printf '#!/usr/bin/env bash\n' + printf 'printf "%%s\\t%%s\\n" "$PWD" "$*" >> %q\n' "$OUT/agent-python.log" + printf 'exec %q "$@"\n' "$real_python" + } > "$shim_dir/$name" + chmod +x "$shim_dir/$name" +done +: > "$OUT/agent-python.log" + +"$APR" --version > "$OUT/apr-version.txt" 2>&1 +sha256sum "$MODEL" | cut -d' ' -f1 > "$OUT/model.sha256" +hostname > "$OUT/hostname.txt" +date -u +%FT%TZ > "$OUT/started.txt" + +prompt="$(cat "$FIXTURE/task.txt")" + +# Every GPU call takes the fleet lock and runs choom'd to 1000, so this run and +# never a CI job is the OOM victim: through gpu-q when asked, else a bounded +# flock. The lock-acquired marker, written once the lock is held, separates +# "never got the GPU" from "apr exited 75". +gate=(flock -w "$LOCK_WAIT" -E 75 "$GPU_LOCK" choom -n 1000 --) +if [ -n "$GPU_Q_PRIO" ]; then + caps=$'\n'"$(gpu-q --caps 2>/dev/null || true)"$'\n' + if [[ "$caps" == *$'\n'wait$'\n'* ]]; then + gate=(env GPUQ_WAIT="$LOCK_WAIT" GPUQ_LOCK="$GPU_LOCK" gpu-q --prio "$GPU_Q_PRIO" --) + else + printf 'note: gpu-q --caps lists no `wait`, so its wait is unbounded; using the bounded flock\n' >&2 + fi +fi +set +e +( + cd "$OUT/project" && + "${gate[@]}" \ + bash -c 'date -u +%FT%TZ > "$1"; shift; exec "$@"' _ "$OUT/lock-acquired" \ + env APR_BIN="$wrapper" PATH="$shim_dir:$PATH" timeout "$TIMEOUT" \ + "$APR" code -p \ + --model "$MODEL" \ + --project "$OUT/project" \ + --output-format json \ + --emit-trace "$OUT/trace.jsonl" \ + --max-turns "$MAX_TURNS" \ + -- "$prompt" +) > "$OUT/stdout.json" 2> "$OUT/stderr.txt" +rc=$? +set -e +date -u +%FT%TZ > "$OUT/finished.txt" + +# The independent re-run: the agent's report of the test is not the test. +set +e +(cd "$OUT/project" && timeout 120 "$real_python" -m unittest test_stats) > "$OUT/unittest.txt" 2>&1 +test_rc=$? +set -e + +python3 "$JUDGE" \ + --out "$OUT" \ + --fixture "$FIXTURE" \ + --model "$MODEL" \ + --host "$HOST" \ + --rc "$rc" \ + --test-rc "$test_rc" \ + --lock-wait "$LOCK_WAIT" \ + --timeout "$TIMEOUT" diff --git a/scripts/check_apr_code_edit_verify.sh b/scripts/check_apr_code_edit_verify.sh new file mode 100755 index 0000000000..81fc6d4fed --- /dev/null +++ b/scripts/check_apr_code_edit_verify.sh @@ -0,0 +1,281 @@ +#!/usr/bin/env bash +# check_apr_code_edit_verify.sh - case table for the #3719 `apr code` judge +# (scripts/lib/apr_code_edit_verify.py) and harness +# (scripts/apr_code_edit_verify.sh). No model, no GPU, no inputs. +# +# The harness rows (h-*) run the real harness end to end against a fake `apr` +# and a scratch lock, never the fleet lock. aprender-62 (#3712): the release +# ladder calls the harness OUTSIDE its own lock, so "every GPU apr call is +# locked" rests on the harness. The rows prove its apr call runs with the lock +# held and oom_score_adj 1000, in both gate modes, and that a held lock gives a +# DECLINE within the bound, never a hang. +# +# Each row builds a synthetic artifact directory in the shape +# scripts/apr_code_edit_verify.sh leaves behind and asserts the verdict and +# the FIRST failing mechanism the judge names. Two rows are real failures the +# judge once got wrong: +# +# zero - the serve child printed `CUDA optimized model ready` for a +# model with `Model ready: 0 layers`, then answered HTTP 500 +# (Qwen3.5-4B on lambda, apr 0.69.0 856009cc9, #3571 step 1). +# The judge called it backend=cuda / "tool call not parsed". +# no-test - the --emit-trace file holds one text block and never a tool +# call, so a judge reading tool_use blocks from it could never +# pass; "the agent ran the test" now comes from python shims. +# +# Exit 0 = every row got the verdict and mechanism it expects; 1 = a row did +# not (named); 2 = python3 missing. +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +FIXTURE="$ROOT/tests/fixtures/apr-code-edit-verify" +JUDGE="$ROOT/scripts/lib/apr_code_edit_verify.py" + +if ! command -v python3 >/dev/null 2>&1; then + printf 'DECLINE python3 not found\n' >&2 + exit 2 +fi + +WORK="$(mktemp -d)" +trap 'rm -rf "${WORK:?}"' EXIT + +LEGACY=$'Model ready: 36 layers, vocab_size=248320, hidden_dim=2560\ngpu-layers: requested=all resolved=36 total=36 (backend=cuda)\nCUDA optimized model ready' +HYBRID=$'Model ready: Qwen3.5 hybrid, 32 layers resident on the GPU, declared context 262144 tokens\nchat template: Qwen3NoThink (thinking off)' +FIXED_LINE='return sum(values) / len(values)' +BUGGY_LINE='return sum(values) / (len(values) - 1)' +FAILED=0 + +# make_case NAME CHILD_STDOUT - a run that did everything right +make_case() { + local d="$WORK/$1" + mkdir -p "$d" + cp -R "$FIXTURE/project" "$d/project" + printf 'abc\n' > "$d/model.sha256" + printf 'apr 0.69.0 (case)\n' > "$d/apr-version.txt" + : > "$d/lock-acquired" + printf 'apr serve ready (3.1s)\n' > "$d/stderr.txt" + printf '%s\n' "$2" > "$d/serve-child.stdout" + : > "$d/serve-child.stderr" + sed -i "s|$BUGGY_LINE|$FIXED_LINE|" "$d/project/stats.py" + printf '%s\t-m unittest test_stats -v\n' "$d/project" > "$d/agent-python.log" + set_result "$1" 'Fixed the denominator; all tests pass.' +} + +set_result() { + python3 -c 'import json,sys; print(json.dumps({"type": "result", "subtype": "success", "result": sys.argv[1], "session_id": "case", "duration_ms": 1}))' "$2" > "$WORK/$1/stdout.json" +} + +unfix() { cp "$FIXTURE/project/stats.py" "$WORK/$1/project/stats.py"; } + +# expect NAME VERDICT MECHANISM_SUBSTRING [APR_RC] +expect() { + local d="$WORK/$1" test_rc=0 line + (cd "$d/project" && python3 -m unittest test_stats >/dev/null 2>&1) || test_rc=$? + line=$(python3 "$JUDGE" --out "$d" --fixture "$FIXTURE" --model /m/Qwen3.5-4B-Q4_K_M.gguf \ + --host case --rc "${4:-0}" --test-rc "$test_rc" --lock-wait 5 --timeout 9 | head -1) || true + if [[ "$line" == "$2"* && "$line" == *"$3"* ]]; then + printf 'ok %s\n' "$1" + else + printf 'FAIL %s: want %s / %s, got: %s\n' "$1" "$2" "$3" "$line" + FAILED=1 + fi +} + +make_case pass "$LEGACY" +expect pass PASS "task completed" + +make_case hybrid "$HYBRID" +printf 'prompt_tokens=4380\n' >> "$WORK/hybrid/serve-child.stderr" +expect hybrid PASS "task completed" +if python3 -c 'import json,sys; c=json.load(open(sys.argv[1])); sys.exit(0 if (c["thinking"], c["context"], c["prompt_tokens"], c["backend"]) == ("off", "4k", 4380, "cuda") else 1)' "$WORK/hybrid/cell.json"; then + printf 'ok hybrid row keys: thinking=off context=4k prompt_tokens=4380 backend=cuda\n' +else + printf 'FAIL hybrid row keys: %s\n' "$(cat "$WORK/hybrid/cell.json")" + FAILED=1 +fi + +make_case pass-keys "$LEGACY" +expect pass-keys PASS "task completed" +if python3 -c 'import json,sys; c=json.load(open(sys.argv[1])); sys.exit(0 if (c["thinking"], c["context"], c["prompt_tokens"]) == ("unknown", "task", None) else 1)' "$WORK/pass-keys/cell.json"; then + printf 'ok unmeasured row keys: thinking=unknown context=task prompt_tokens=null\n' +else + printf 'FAIL unmeasured row keys: %s\n' "$(cat "$WORK/pass-keys/cell.json")" + FAILED=1 +fi + +make_case model-choice "${HYBRID/thinking off/thinking the model\'s choice}" +expect model-choice PASS "task completed" +if python3 -c 'import json,sys; c=json.load(open(sys.argv[1])); sys.exit(0 if (c["thinking"], c["thinking_raw"]) == ("unknown", "the model'"'"'s choice") else 1)' "$WORK/model-choice/cell.json"; then + printf 'ok a thinking value other than on/off keys onto no cell: thinking=unknown\n' +else + printf 'FAIL model-choice row keys: %s\n' "$(cat "$WORK/model-choice/cell.json")" + FAILED=1 +fi + +make_case hybrid-cpu "${HYBRID/GPU/CPU}" +expect hybrid-cpu FAIL "fell back to CPU" + +make_case zero $'Model ready: 0 layers, vocab_size=248320, hidden_dim=2560\ngpu-layers: requested=all resolved=0 total=0 (backend=cuda)\nCUDA optimized model ready' +: > "$WORK/zero/stdout.json" +: > "$WORK/zero/agent-python.log" +unfix zero +printf '%s\n' 'Error: driver error: network error: apr serve HTTP 500: {"error":"Model architecture not supported for GPU-resident path"}' >> "$WORK/zero/stderr.txt" +expect zero FAIL "serve child did not load: the child reported ready with 0 layers" 1 + +make_case cpu "$LEGACY" +printf '[GPU->CPU FALLBACK] out of memory\n' > "$WORK/cpu/serve-child.stderr" +expect cpu FAIL "fell back to CPU: serve child printed a CPU path" + +make_case no-load "" +printf 'Error: apr serve exited\n' > "$WORK/no-load/stderr.txt" +expect no-load FAIL "serve child did not load: the driver never reported" 1 + +make_case partial "${LEGACY/resolved=36/resolved=20}" +expect partial FAIL "never showed every layer resident on CUDA" + +make_case refused "$LEGACY" +: > "$WORK/refused/stdout.json" +printf '%s\n' 'Error: driver error: network error: apr serve HTTP 500: {"error":"boom"}' >> "$WORK/refused/stderr.txt" +expect refused FAIL "serve child refused the forward" 1 + +make_case unparsed "$LEGACY" +unfix unparsed +: > "$WORK/unparsed/agent-python.log" +set_result unparsed $'I will fix it.\n\n{"name": "file_edit", "input": {}}\n' +expect unparsed FAIL "tool call not parsed" + +make_case no-edit "$LEGACY" +unfix no-edit +set_result no-edit '' +printf 'gpu-q: waiting (prio 1, 2/5 in queue)\n%s\n' "$(cat "$WORK/no-edit/stderr.txt")" > "$WORK/no-edit/stderr.txt" +expect no-edit FAIL "wrong edit: stats.py was not changed" +if python3 -c 'import json,sys; c=json.load(open(sys.argv[1])); sys.exit(1 if "gpu-q:" in c["evidence"] else 0)' "$WORK/no-edit/cell.json"; then + printf 'ok gpu-q queue lines are not quoted as evidence\n' +else + printf 'FAIL no-edit evidence quotes the gpu-q queue: %s\n' "$(cat "$WORK/no-edit/cell.json")" + FAILED=1 +fi + +make_case test-edited "$LEGACY" +printf '# edited\n' >> "$WORK/test-edited/project/test_stats.py" +expect test-edited FAIL "wrong edit: test_stats.py was modified" + +make_case two-lines "$LEGACY" +sed -i 's|"""Small statistics helpers."""|"""Stats."""|' "$WORK/two-lines/project/stats.py" +expect two-lines FAIL "wrong edit: stats.py changed by more than one line" + +make_case still-fails "$LEGACY" +unfix still-fails +sed -i 's|(len(values) - 1)|(len(values) + 1)|' "$WORK/still-fails/project/stats.py" +expect still-fails FAIL "independent test re-run still fails" + +make_case extra-file "$LEGACY" +: > "$WORK/extra-file/project/notes.txt" +expect extra-file FAIL "wrong edit: files added" + +make_case no-test "$LEGACY" +: > "$WORK/no-test/agent-python.log" +expect no-test FAIL "test not run" + +make_case answer "$LEGACY" +set_result answer 'I changed a line.' +expect answer FAIL "wrong final answer: the final answer does not report" + +make_case no-envelope "$LEGACY" +: > "$WORK/no-envelope/stdout.json" +expect no-envelope FAIL "wrong final answer: no" + +make_case lock "$LEGACY" +rm "$WORK/lock/lock-acquired" +expect lock DECLINE "gpu lock not acquired" 75 + +# ---- harness rows: the real harness, a fake apr, a scratch lock ------------- +HARNESS="$ROOT/scripts/apr_code_edit_verify.sh" +BIN="$WORK/bin" +mkdir -p "$BIN" +HEAD_SHA=$(git -C "$ROOT" rev-parse --short HEAD 2>/dev/null || printf 'nogit') +# The fake answers --version with HEAD's sha (scripts/apr_bin.sh checks it), +# plays the serve child through the harness's APR_BIN wrapper, records whether +# the GPU lock is held and its own oom_score_adj, then does the task right. +cat > "$BIN/fake-apr" < "\$FAKE_PROBE" + "\$APR_BIN" serve --fake-child > /dev/null + printf 'apr serve ready (0.1s)\n' >&2 + sed -i 's|$BUGGY_LINE|$FIXED_LINE|' stats.py + python3 -m unittest test_stats > /dev/null 2>&1 + printf '{"type": "result", "subtype": "success", "result": "Fixed the denominator; all tests pass."}\n' ;; +esac +FAKE +# The stub keeps gpu-q v3's contract: `--caps` lists prio and wait, GPUQ_WAIT +# bounds the wait with exit 75, then it takes the lock and choom and runs the +# job. OLD is a gpu-q without `wait`, whose wait has no bound at all. +cat > "$BIN/gpu-q" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "--caps" ]; then printf 'prio\nwait\n'; exit 0; fi +[ "${1:-}" = "--prio" ] && shift 2 +[ "${1:-}" = "--" ] && shift +if [ -n "${GPUQ_WAIT:-}" ]; then + exec flock -w "$GPUQ_WAIT" -E 75 "${GPUQ_LOCK:?}" choom -n 1000 -- "$@" +fi +exec flock "${GPUQ_LOCK:?}" choom -n 1000 -- "$@" +STUB +OLD="$WORK/old-gpu-q" +mkdir -p "$OLD" +cat > "$OLD/gpu-q" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "--caps" ]; then exit 2; fi +[ "${1:-}" = "--prio" ] && shift 2 +[ "${1:-}" = "--" ] && shift +exec flock "${GPUQ_LOCK:?}" choom -n 1000 -- "$@" +STUB +chmod +x "$BIN/fake-apr" "$BIN/gpu-q" "$OLD/gpu-q" +printf 'not a model\n' > "$WORK/model.gguf" + +# harness_row NAME WANT_RC WANT_PROBE MAX_SECONDS [harness args...] +# A harness that hangs (a double lock, a lost bound) is killed and fails the row. +harness_row() { + local name="$1" want_rc="$2" want_probe="$3" max_s="$4" rc=0 t0 took probe + shift 4 + t0=$(date +%s) + (cd "$ROOT" && PATH="${ROW_PATH:-$BIN}:$PATH" APR_BIN="$BIN/fake-apr" APR_GPU_LOCK="$WORK/gpu.lock" \ + FAKE_PROBE="$WORK/$name.probe" \ + timeout "$((max_s + 5))" bash "$HARNESS" --model "$WORK/model.gguf" --host case --out "$WORK/$name" "$@") \ + > "$WORK/$name.log" 2>&1 || rc=$? + took=$(( $(date +%s) - t0 )) + probe=$(cat "$WORK/$name.probe" 2>/dev/null || printf 'never-ran') + if [ "$rc" = "$want_rc" ] && [ "$probe" = "$want_probe" ] && [ "$took" -le "$max_s" ]; then + printf 'ok %s (rc=%s, %s, %ss)\n' "$name" "$rc" "$probe" "$took" + else + printf 'FAIL %s: want rc=%s probe=%s within %ss, got rc=%s probe=%s in %ss\n' \ + "$name" "$want_rc" "$want_probe" "$max_s" "$rc" "$probe" "$took" + sed 's/^/ /' "$WORK/$name.log" | tail -5 + FAILED=1 + fi +} + +harness_row h-flock-free 0 "held=yes oom=1000" 60 +harness_row h-gpuq-free 0 "held=yes oom=1000" 60 --gpu-q 1 +ROW_PATH="$OLD:$BIN" harness_row h-old-gpuq-free 0 "held=yes oom=1000" 60 --gpu-q 1 +flock "$WORK/gpu.lock" sleep 60 & +holder=$! +sleep 0.5 +harness_row h-flock-held 2 never-ran 30 --lock-wait 2 +harness_row h-gpuq-held 2 never-ran 30 --gpu-q 1 --lock-wait 2 +ROW_PATH="$OLD:$BIN" harness_row h-old-gpuq-held 2 never-ran 30 --gpu-q 1 --lock-wait 2 +kill "$holder" 2>/dev/null || true +wait "$holder" 2>/dev/null || true + +if [ "$FAILED" -ne 0 ]; then + printf 'FAIL: the apr code judge or harness got a row wrong\n' + exit 1 +fi +printf 'OK: every row got its verdict and first failing mechanism\n' diff --git a/scripts/lib/apr_code_edit_verify.py b/scripts/lib/apr_code_edit_verify.py new file mode 100644 index 0000000000..ca3f1dcd96 --- /dev/null +++ b/scripts/lib/apr_code_edit_verify.py @@ -0,0 +1,341 @@ +#!/usr/bin/env python3 +"""Judge one `apr code` edit-and-verify run from its artifacts (#3719). + +scripts/apr_code_edit_verify.sh runs the agent and leaves an artifact +directory behind. This module reads only those artifacts: the working copy, +the independent test re-run, the python-shim log of what the agent's shell +ran, the final answer, and the serve child's own output. It writes +/cell.json and prints one summary line. + +The mechanisms are checked in pipeline order and the FIRST one that fails is +the cell's mechanism, using the names from #3719 done_when 1: + + serve child did not load -> fell back to CPU -> (serve child refused the + forward) -> tool call not parsed -> wrong edit -> test not run + -> wrong final answer + +"serve child refused the forward" is not in #3719's list: it names a child +that loaded every layer on CUDA and then answered the completion with an HTTP +error, which none of the listed names describes. + +What each mechanism reads, and why: + * The --emit-trace file is NOT evidence of tool calls. It holds exactly four + records and the assistant turn is one text block, whatever the agent did + (crates/aprender-orchestrate/src/agent/code.rs emit_ccpa_trace). A judge + reading tool_use blocks from it could never pass. + * "tool call not parsed" needs a call the model emitted and the driver did + not execute: stats.py unchanged while the final answer still carries the + markup the driver parses ( or a ```json block, see + agent/driver/realizar.rs parse_tool_calls). + * "test not run" reads the python shims' log: the agent's shell tool runs + `sh -c` with the inherited PATH, and the shims are first on it. + +Exit: 0 PASS, 1 FAIL, 2 decline (the GPU lock was never acquired). +""" + +import argparse +import difflib +import json +import re +import sys +from pathlib import Path + +ANSI = re.compile(r"\x1b\[[0-9;]*m") + +# What the serve child prints on its CUDA path, and on each way it can end up +# on the CPU instead (crates/apr-cli/src/commands/serve/). +# +# `CUDA optimized model ready` is a LOAD-time line and is not evidence of a CUDA +# forward: measured on 856009cc9, the child printed it for Qwen3.5-4B after +# `Model ready: 0 layers` and `gpu-layers: requested=all resolved=0 total=0`, +# and the first request then failed with HTTP 500 (#3571 step 1). So the +# backend is `cuda` only when every layer is resident on CUDA and a completion +# actually came back; the residency line is the evidence cited. +CUDA_READY = re.compile(r"CUDA optimized model ready") +CPU_MARKERS = re.compile( + r"\[GPU->CPU FALLBACK\]|Using CPU inference|Q4K CPU inference ready" + r"|layers resident on the CPU" +) +# The generic GGUF route: a layer count, then the gpu-layers residency line. +MODEL_LAYERS = re.compile(r"Model ready: (\d+) layers") +GPU_LAYERS = re.compile(r"gpu-layers: requested=\S+ resolved=(\d+) total=(\d+) \(backend=(\w+)\)") +# The Qwen35Session route (#3571 step 2, aprender-c7) states both in one line: +# `Model ready: Qwen3.5 hybrid, N layers resident on the GPU|CPU, declared context C tokens`. +HYBRID_READY = re.compile(r"Model ready: .*?(\d+) layers resident on the (GPU|CPU)") +# The driver prints this once the child answers its health check. +SERVE_READY = re.compile(r"apr serve ready \(") +# The driver's error when the child answers a completion with an HTTP error. +SERVE_HTTP_ERROR = re.compile(r"apr serve HTTP (\d+): (.*)") +# The thinking mode the serve child renders (the Qwen35Session route prints +# `chat template: Qwen3NoThink (thinking off)`). The driver strips +# blocks before parsing, so nothing else can show it. The template name varies +# with the shared selection (#3755), so only the `(thinking ...)` part is read. +# Its value is `off`, `on` or `the model's choice`; only on/off key a ladder +# row, and anything else is kept raw and keys onto no cell. +THINKING_LINE = re.compile(r"chat template:.*\(thinking ([^)]+)\)") +# A per-request prompt size printed by the child. session_end.tokens_in in the +# trace is summed over turns (agent/result.rs accumulate), so it is not one. +PROMPT_TOKENS_LINE = re.compile(r"\bprompt_tokens[=: ]+(\d+)") +# Tool-call markup the driver parses. Left in the final answer, it is a call +# the model made and the driver never executed. +UNPARSED_CALL = re.compile(r"|```json| 0, hybrid + layers_line = first_match(child, MODEL_LAYERS) + layers = int(MODEL_LAYERS.search(layers_line).group(1)) if layers_line else None + gpu_line = first_match(child, GPU_LAYERS) + if not gpu_line: + return layers, False, layers_line + resolved, total, backend = GPU_LAYERS.search(gpu_line).groups() + resident = (backend == "cuda" and int(resolved) == int(total) > 0 + and bool(CUDA_READY.search(child))) + return layers, resident, "\n".join(x for x in (layers_line, gpu_line) if x) + + +def judge(a): + out = Path(a.out) + fixture = Path(a.fixture) / "project" + work = out / "project" + # gpu-q prints its queue position to stderr while it waits; those lines are + # the queue talking, not apr, and are kept out of every evidence excerpt. + stderr = "\n".join(ln for ln in read(out / "stderr.txt").splitlines() + if not ln.startswith("gpu-q: ")) + stdout = read(out / "stdout.json") + child = read(out / "serve-child.stdout") + "\n" + read(out / "serve-child.stderr") + + cell = { + "verb": "code", + "host": a.host, + "file": Path(a.model).name, + "sha256": read(out / "model.sha256").strip(), + "apr_version": read(out / "apr-version.txt").strip(), + "hostname": read(out / "hostname.txt").strip(), + "started": read(out / "started.txt").strip(), + "finished": read(out / "finished.txt").strip(), + "rc": a.rc, + "test_rc": a.test_rc, + "task": "tests/fixtures/apr-code-edit-verify", + } + + if not (out / "lock-acquired").exists(): + cell.update(verdict="decline", mechanism="gpu lock not acquired", + reason=f"/tmp/apr-gpu.lock not acquired within {a.lock_wait}s; nothing ran", + backend="none", fallback=False, evidence="") + return cell, 2 + + timed_out = a.rc == 124 + cpu_line = first_match(child, CPU_MARKERS) + layers, resident, residency_line = residency(child) + http_error = SERVE_HTTP_ERROR.search(stderr) + env = envelope(stdout) + result = str(env.get("result", "")) if env else "" + answered = env is not None + + if cpu_line: + backend = "cpu" + elif resident and answered: + backend = "cuda" + else: + backend = "unknown" + cell.update(backend=backend, fallback=bool(cpu_line), + evidence=cpu_line or (residency_line if backend == "cuda" else ""), + model_layers=layers, answer_chars=len(result)) + + thinking = THINKING_LINE.search(child) + raw = thinking.group(1).strip() if thinking else "" + cell["thinking"] = raw if raw in ("on", "off") else "unknown" + cell["thinking_raw"] = raw + sizes = [int(n) for n in PROMPT_TOKENS_LINE.findall(child)] + cell["prompt_tokens"] = max(sizes) if sizes else None + cell["context"] = "4k" if sizes and max(sizes) >= RUNG_4K_TOKENS else "task" + cell["max_tokens"] = DRIVER_MAX_TOKENS + + calls = [ln for ln in read(out / "agent-python.log").splitlines() if ln.strip()] + cell["agent_python_calls"] = calls[:20] + + def fail(mechanism, reason, evidence=None): + if timed_out: + reason = f"{reason} (apr code timed out after {a.timeout}s)" + cell.update(verdict="fail", mechanism=mechanism, reason=reason) + if evidence is not None: + cell["evidence"] = evidence + return cell, 1 + + # 1. serve child did not load (a model with no layers is not loaded, even + # when the child calls itself ready) + if not SERVE_READY.search(stderr): + return fail("serve child did not load", + "the driver never reported `apr serve ready`", + last_lines(child) or last_lines(stderr)) + if layers == 0: + detail = (f"; first request: HTTP {http_error.group(1)} {http_error.group(2)[:200]}" + if http_error else "") + return fail("serve child did not load", + f"the child reported ready with 0 layers{detail}", residency_line) + + # 2. fell back to CPU, or never showed every layer resident on CUDA + if cpu_line: + return fail("fell back to CPU", "serve child printed a CPU path", cpu_line) + if not resident: + return fail("fell back to CPU", "serve child never showed every layer resident on CUDA", + residency_line or last_lines(child)) + + # 2b. the child loaded on CUDA but refused the completion + if http_error and not answered: + return fail("serve child refused the forward", + f"HTTP {http_error.group(1)}: {http_error.group(2)[:300]}", residency_line) + + # 3. tool call not parsed: no edit landed, and the answer still carries + # the markup of a call the driver should have executed + src_before, src_after = read(fixture / EDITED_FILE), read(work / EDITED_FILE) + unparsed = UNPARSED_CALL.search(result) + if src_after == src_before and unparsed: + return fail("tool call not parsed", + f"{EDITED_FILE} unchanged and the final answer carries `{unparsed.group(0)}`", + result[max(0, unparsed.start() - 80):unparsed.start() + 220]) + + # 4. wrong edit + before_files, after_files = project_files(fixture), project_files(work) + cell["files_added"] = sorted(after_files - before_files) + cell["files_removed"] = sorted(before_files - after_files) + diff = "".join(difflib.unified_diff(src_before.splitlines(True), src_after.splitlines(True), + EDITED_FILE, EDITED_FILE)) + (out / "stats.diff").write_text(diff) + cell["edit"] = [ln for ln in diff.splitlines() + if ln[:1] in "+-" and ln[:3] not in ("+++", "---")] + if read(work / TEST_FILE) != read(fixture / TEST_FILE): + return fail("wrong edit", f"{TEST_FILE} was modified; the task forbids it") + if cell["files_added"] or cell["files_removed"]: + return fail("wrong edit", + f"files added {cell['files_added']} / removed {cell['files_removed']}") + if src_after == src_before: + return fail("wrong edit", f"{EDITED_FILE} was not changed and no unparsed call was left", + result[:300] or last_lines(stderr)) + if not one_line_edit(src_before, src_after): + return fail("wrong edit", f"{EDITED_FILE} changed by more than one line", diff[:600]) + if a.test_rc != 0: + return fail("wrong edit", "one-line edit made, but the independent test re-run still fails", + last_lines(read(out / "unittest.txt"))) + + # 5. test not run (by the agent; the harness's re-run is not in the log) + ran = [c for c in calls if "unittest" in c or "test_stats" in c] + cell["test_call"] = ran[0] if ran else "" + if not ran: + return fail("test not run", "the agent's shell never ran python on the fixture's test", + "; ".join(calls[:3]) or "no python invocation at all") + + # 6. wrong final answer + if env is None or env.get("subtype") != "success": + return fail("wrong final answer", "no `type: result, subtype: success` envelope on stdout", + last_lines(stdout) or last_lines(stderr)) + if not re.search(r"\b(pass(es|ed|ing)?|OK)\b", result, re.IGNORECASE): + return fail("wrong final answer", "the final answer does not report the passing test", + result[:300]) + if a.rc != 0: + return fail("wrong final answer", f"task done but apr code exited {a.rc}") + + cell.update(verdict="pass", mechanism="", reason="") + return cell, 0 + + +def main(): + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--out", required=True) + p.add_argument("--fixture", required=True) + p.add_argument("--model", required=True) + p.add_argument("--host", required=True) + p.add_argument("--rc", type=int, required=True) + p.add_argument("--test-rc", type=int, required=True) + p.add_argument("--lock-wait", type=int, default=0) + p.add_argument("--timeout", type=int, default=0) + a = p.parse_args() + cell, code = judge(a) + Path(a.out, "cell.json").write_text(json.dumps(cell, indent=1) + "\n") + why = f"{cell['mechanism']}: {cell['reason']}" if cell.get("mechanism") else "task completed" + print(f"{cell['verdict'].upper():7} {cell['host']:7} {cell['file']} " + f"backend={cell['backend']} {why}") + if cell.get("evidence"): + print(f" evidence: {cell['evidence'].splitlines()[0][:200]}") + return code + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/fixtures/apr-code-edit-verify/project/stats.py b/tests/fixtures/apr-code-edit-verify/project/stats.py new file mode 100644 index 0000000000..72ab447160 --- /dev/null +++ b/tests/fixtures/apr-code-edit-verify/project/stats.py @@ -0,0 +1,15 @@ +"""Small statistics helpers.""" + + +def mean(values): + """Arithmetic mean of a non-empty list of numbers.""" + return sum(values) / (len(values) - 1) + + +def median(values): + """Middle value of a non-empty list of numbers.""" + ordered = sorted(values) + mid = len(ordered) // 2 + if len(ordered) % 2 == 1: + return ordered[mid] + return (ordered[mid - 1] + ordered[mid]) / 2 diff --git a/tests/fixtures/apr-code-edit-verify/project/test_stats.py b/tests/fixtures/apr-code-edit-verify/project/test_stats.py new file mode 100644 index 0000000000..227453afff --- /dev/null +++ b/tests/fixtures/apr-code-edit-verify/project/test_stats.py @@ -0,0 +1,16 @@ +import unittest + +from stats import mean, median + + +class TestStats(unittest.TestCase): + def test_mean(self): + self.assertEqual(mean([2, 4, 6]), 4) + + def test_median(self): + self.assertEqual(median([3, 1, 2]), 2) + self.assertEqual(median([4, 1, 3, 2]), 2.5) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/fixtures/apr-code-edit-verify/task.txt b/tests/fixtures/apr-code-edit-verify/task.txt new file mode 100644 index 0000000000..bd23adf5ba --- /dev/null +++ b/tests/fixtures/apr-code-edit-verify/task.txt @@ -0,0 +1 @@ +The unit test test_mean in test_stats.py fails. Find the bug in stats.py and fix it with a one-line edit. Do not change test_stats.py. Then run `python3 -m unittest test_stats -v` in the project directory and report whether all tests pass.