From 86215ce565136e48aa0799062081ed8d4cab287c Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 13:51:34 +0200 Subject: [PATCH 1/7] build: generate header prerequisites with -MMD -MP (#1741) Every engine and test rule listed its headers by hand on one line, the most edited line in the Makefile and a merge conflict for every open PR touching the same rule. The lists also drifted: many built targets on dev read headers their rule does not name (measured in the PR), so editing one of those leaves a stale binary while make reports success. -MMD -MP in CFLAGS now writes .d beside each binary and object, and the Makefile includes those files. Header lists are removed from 166 rules (9 engines, 154 tests, 3 objects). Three cases the compiler cannot cover on its own: - A command compiling several .c files writes the last unit's deps only. Eight such tests keep their lists (AUTODEP_MULTI_TU). With CUDA/HIP, qwen36_tier.c joined 14 command lines the same way; it is now its own object, qwen36_tier.o, and QWEN36_TIER_SRC is renamed QWEN36_TIER_OBJ. - Only 6 of the 166 rules depend on .build-config, so the CFLAGS change would not rebuild the rest and they would have no .d. Each target therefore depends on its own .d through an empty rule: a missing .d makes it out of date, and the rebuild writes one. - Rules built through Makefile.deepseek-v4 or the adapter targets (AUTODEP_ELSEWHERE), and the nvcc, MSVC-hosted CUDA and Metal objects, keep their lists; nvcc never receives CFLAGS. test_makefile_deps.py now checks that every rule in AUTODEP_BINS compiles one unit with -MMD in the default and the GPU configuration, that no such rule lists headers by hand, that every built target's .d covers its source's unconditional includes, and that make -q -W on a header from a .d answers "rebuild". clean.py removes *.d; c/.gitignore ignores them. test_glm53_metal_source.py no longer expects backend_metal.h in the glm53 rule. --- c/.gitignore | 2 + c/Makefile | 421 ++++++++++++++++------------ c/tests/test_glm53_metal_source.py | 3 +- c/tests/test_makefile_deps.py | 432 +++++++++++++++++++++-------- c/tools/clean.py | 8 +- 5 files changed, 561 insertions(+), 305 deletions(-) diff --git a/c/.gitignore b/c/.gitignore index b3da9261a..d8c4d6603 100644 --- a/c/.gitignore +++ b/c/.gitignore @@ -44,6 +44,8 @@ tests/test_kimi_serve_framing tests/test_v4_serve_framing tests/test_inkling_serve_framing *.o +# dependency files -MMD writes beside each binary and object (#1741) +*.d *.dll cublasLt64*.dll cudart64*.dll diff --git a/c/Makefile b/c/Makefile index 370160e7f..4aaf6e6c6 100644 --- a/c/Makefile +++ b/c/Makefile @@ -220,6 +220,14 @@ EXE = endif endif +# Header prerequisites are generated by the compiler, not listed by hand +# (#1741). -MMD writes .d beside every binary or object compiled with +# CFLAGS, naming each project header that compile actually read; -MP adds an +# empty rule per header, so deleting one does not break the next build. The +# .d files are read at the end of this Makefile, next to the list of rules they +# cover (AUTODEP_BINS). +CFLAGS += -MMD -MP + # --- install --- PREFIX ?= /usr/local BINDIR ?= $(PREFIX)/bin @@ -811,18 +819,18 @@ endif endif .build-config: ; -colibri$(EXE): colibri.c sse41_kernels.h exact_dot.h oracle.h pin_pool.h cli_args.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h omp_tune.h kv_fp8.h kv_tq.h abl.h backend_cuda.h backend_metal.h backend_vulkan.h backend_xdna.h decode_batch.h edge_adapters.h edge_runtime.h edge_tok_internal.h evidence_digest.h schema_gbnf.h segment_adapter_internal.h segment_adapters.h segment_runtime.h tier.h $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) $(XDNA_OBJ) .build-config +colibri$(EXE): colibri.c $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) $(XDNA_OBJ) .build-config $(CC) $(CFLAGS) colibri.c $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(XDNA_OBJ) -o colibri$(EXE) $(LDFLAGS) # Vulkan backend object (plain C + vulkan headers) and its SPIR-V shaders. -backend_vulkan.o: backend_vulkan.c backend_vulkan.h .build-config +backend_vulkan.o: backend_vulkan.c .build-config $(CC) $(CFLAGS) -c backend_vulkan.c -o $@ shaders/%.spv: shaders/%.comp @command -v $(GLSLC) >/dev/null 2>&1 || { echo "glslc not found: install shaderc (for VK=1)" >&2; exit 1; } $(GLSLC) --target-env=vulkan1.2 $< -o $@ # Windows runtime loader object: resolves coli_cuda_* from coli_cuda.dll. -backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config +backend_loader.o: backend_loader.c .build-config $(CC) $(CFLAGS) -c backend_loader.c -o $@ # Optional XDNA2 (Ryzen AI NPU) lane, host side only. This object resolves an @@ -836,7 +844,7 @@ backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config # side is caught without building a whole host. .PHONY: xdna-obj xdna-obj: backend_xdna.o -backend_xdna.o: backend_xdna.c backend_xdna.h .build-config +backend_xdna.o: backend_xdna.c .build-config $(CC) $(CFLAGS) -c backend_xdna.c -o $@ # Windows CUDA DLL: compile backend_cuda.cu with nvcc (+MSVC cl.exe as host @@ -1108,10 +1116,10 @@ dsv4-cuda-test: tests/test_dsv4_dense_batch_cuda.c tests/test_dsv4_attention_bat ./dsv4_moe_cuda_test$(EXE) # Pure-CPU unit tests for the two headers the kernels consume. No GPU, no model. -tests/test_dsv4_mhc$(EXE): tests/test_dsv4_mhc.c dsv4_mhc.h +tests/test_dsv4_mhc$(EXE): tests/test_dsv4_mhc.c $(CC) $(CFLAGS) tests/test_dsv4_mhc.c -o tests/test_dsv4_mhc$(EXE) $(LDFLAGS) -tests/test_dsv4_quant$(EXE): tests/test_dsv4_quant.c dsv4_quant.h +tests/test_dsv4_quant$(EXE): tests/test_dsv4_quant.c $(CC) $(CFLAGS) tests/test_dsv4_quant.c -o tests/test_dsv4_quant$(EXE) $(LDFLAGS) # convenience alias: kernel correctness on AMD (same test, hipcc toolchain) hip-test: @@ -1151,7 +1159,7 @@ tests/bench_cuda_resident_batch$(EXE): tests/bench_cuda_resident_batch.cu backen # the GPU sits idle. Same guard #783 put on kimi_k3, which since gaining an # MXFP4 expert path no longer needs it. tests/test_makefile_cuda_scope.py # asserts this shape. -olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h kv_prefix.h pin_pool.h route_trace.h serve_codec.h serve_budget.h edge_adapters.h edge_runtime.h edge_tok_internal.h fused_simd.h segment_adapter_internal.h segment_adapters.h segment_runtime.h sse41_kernels.h +olmoe$(EXE): olmoe.c $(CC) $(NOCUDA_CFLAGS) olmoe.c -o olmoe$(EXE) $(NOCUDA_LDFLAGS) # Qwen3.6-35B-A3B engine (hybrid Gated Attention + Gated DeltaNet + streaming @@ -1169,14 +1177,21 @@ olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode. # the direct link, or `make qwen36.exe CUDA_DLL=1` silently produced a CPU-only # engine (#1533). ifneq (,$(filter 1,$(CUDA) $(CUDA_DLL) $(HIP) $(HIP_DLL))) -QWEN36_TIER_SRC = qwen36_tier.c +QWEN36_TIER_OBJ = qwen36_tier.o QWEN36_CFLAGS = $(CFLAGS) QWEN36_LDFLAGS = $(LDFLAGS) else -QWEN36_TIER_SRC = +QWEN36_TIER_OBJ = QWEN36_CFLAGS = $(NOCUDA_CFLAGS) QWEN36_LDFLAGS = $(NOCUDA_LDFLAGS) endif +# The tier is its own object rather than a second source on each engine's +# command line: GCC writes a .d for only the LAST unit of a multi-source +# compile, so qwen36.c's headers would drop out of qwen36.d whenever the tier +# is built in (#1741). Same $(CFLAGS) the one-command build gave it: in this +# configuration QWEN36_CFLAGS is $(CFLAGS). +qwen36_tier.o: qwen36_tier.c .build-config + $(CC) $(CFLAGS) -c qwen36_tier.c -o $@ # On Windows, EXE=.exe. Keep a bare qwen36 target so GNU make does not # fall through to its implicit %: %.c rule and compile qwen36.c alone. ifneq ($(EXE),) @@ -1185,22 +1200,22 @@ qwen36: qwen36$(EXE) endif # Rebuild when CUDA_DLL changes; otherwise an existing CPU-only executable can # be reported as up to date despite selecting the Windows CUDA DLL tier. -qwen36$(EXE): qwen36.c sse41_kernels.h decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) .build-config - $(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS) +qwen36$(EXE): qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) .build-config + $(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS) # DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts # stream from the official checkpoint and quant.h's mxfp4 kernel reads them as they # are, so there is no conversion target here to go with it. -deepseek_v41$(EXE): deepseek_v41.c sse41_kernels.h cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h \ - sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h .build-config +deepseek_v41$(EXE): deepseek_v41.c \ + .build-config $(CC) $(CFLAGS) deepseek_v41.c -o deepseek_v41$(EXE) $(LDFLAGS) # Qwen3.8-Flash-Next text-only sibling. qwen38.c owns the direct checkpoint # loader and SERVE=1 protocol. With CUDA=1 it links the same expert tier as # qwen36 (fp8 streaming mode: hot experts get VRAM copies, the RAM LRU stays); # without it the tier header's inline stubs keep the build toolkit-free. -qwen38$(EXE): qwen38.c sse41_kernels.h pin_pool.h cli_args.h qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h omp_tune.h quant.h fp8_format.h idot.h route_trace.h tok.h tok_unicode.h tok_unicode_o200k.h serve_codec.h edge_adapter_internal.h edge_adapters.h edge_runtime.h qwen38_vision.h segment_adapter_internal.h segment_adapters.h segment_runtime.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) qwen38.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen38$(EXE) $(QWEN36_LDFLAGS) +qwen38$(EXE): qwen38.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) qwen38.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o qwen38$(EXE) $(QWEN36_LDFLAGS) .PHONY: qwen38-tiny-generate qwen38-tiny-check qwen38-tiny-generate: @@ -1254,51 +1269,51 @@ qwen38-tier-engine-check: qwen38-tiny-fp8-generate tests/test_qwen38_tier_engine # Same tier sources as the engine: the test includes qwen36.c, so with CUDA=1 # it needs qwen36_tier.c and the backend object too (without CUDA, the header's # inline stubs cover it and both are empty). -tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS) +tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS) # CACHE_ROUTE: route_select() against a residency table, no model needed. -tests/test_qwen36_cache_route$(EXE): tests/test_qwen36_cache_route.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS) +tests/test_qwen36_cache_route$(EXE): tests/test_qwen36_cache_route.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS) -tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # Both tokenizer.json merge spellings ("a b" strings and ["a","b"] pairs) # must index the same merge table; the pair form is what Qwen3.6 ships. -tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # A byte-counted serving payload may end mid-character; the pre-tokenizer must # not read past it. qwen38 already gates this (tests/test_qwen38_tokenizer.c). -tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # push_id / bpe_piece must refuse loudly, not crash, on a failed realloc during growth. -tests/test_qwen36_encode_oom$(EXE): tests/test_qwen36_encode_oom.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_encode_oom$(EXE): tests/test_qwen36_encode_oom.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # slot_ensure_int8's wiring onto #1271's unpack_int4_to_int8: regression check # that its output matches the old separate scalar nibble-unpack loop it replaced. -tests/test_qwen36_slot_int8$(EXE): tests/test_qwen36_slot_int8.c qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_slot_int8$(EXE): tests/test_qwen36_slot_int8.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # the dense trunk's integer path: quantizer contract, int4 planar packing, dispatch. -tests/test_qwen36_dense_idot$(EXE): tests/test_qwen36_dense_idot.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_dense_idot$(EXE): tests/test_qwen36_dense_idot.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # #1653: added tokens are split out before the regex pre-tokenizer, HF-style. -tests/test_qwen36_tokenizer$(EXE): tests/test_qwen36_tokenizer.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_tokenizer$(EXE): tests/test_qwen36_tokenizer.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # Reproducible local timing evidence; intentionally not a noisy CI perf gate. -tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -inkling$(EXE): inkling.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h serve_budget.h backend_cuda_ink.h backend_metal.h edge_adapters.h edge_runtime.h edge_tok_internal.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(INK_CUDA_OBJ) $(METAL_OBJ) +inkling$(EXE): inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) -o inkling$(EXE) $(LDFLAGS) # ENGINES WITHOUT A CUDA BACKEND (#783). @@ -1315,10 +1330,10 @@ NOCUDA_LDFLAGS = $(filter-out -lcudart -lstdc++ -lcuda -L$(CUDA_HOME)/lib64 \ # GLM-5.3-Flash: routed experts stream from the int4-gs64 container. # METAL=1 accelerates resident matrices and routed MoE; CPU remains fallback. -glm53$(EXE): glm53.c sse41_kernels.h decode_batch.h pin_pool.h cli_args.h st.h json.h stop_ids.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h serve_poll.h route_trace.h quant.h fp8_format.h idot.h hyper_connections.h delta_attention.h sparse_index.h vision_tower.h backend_metal.h backend_vulkan.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) +glm53$(EXE): glm53.c $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) $(CC) $(CFLAGS) glm53.c $(METAL_OBJ) $(VK_OBJ) -o glm53$(EXE) $(LDFLAGS) -kimi_k3$(EXE): kimi_k3.c sse41_kernels.h serve_budget.h cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda.h backend_metal.h backend_vulkan.h edge_adapters.h edge_runtime.h edge_tok_internal.h hybrid_split.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ) +kimi_k3$(EXE): kimi_k3.c $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ) $(CC) $(CFLAGS) kimi_k3.c $(CUDA_OBJ) $(VK_OBJ) $(METAL_OBJ) -o kimi_k3$(EXE) $(LDFLAGS) # Use a baseline that matches the compiler target. macOS already targets a @@ -1340,26 +1355,26 @@ endif portable: $(MAKE) colibri$(EXE) ARCH=$(PORTABLE_ARCH) -iobench$(EXE): iobench.c compat.h +iobench$(EXE): iobench.c $(CC) $(CFLAGS) iobench.c -o iobench$(EXE) $(LDFLAGS) -tests/test_serve_sentinel$(EXE): tests/test_serve_sentinel.c compat.h serve_codec.h +tests/test_serve_sentinel$(EXE): tests/test_serve_sentinel.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_ue8m0$(EXE): tests/test_ue8m0.c st.h json.h compat.h +tests/test_ue8m0$(EXE): tests/test_ue8m0.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Compiles the engine with -DCOLI_XDNA so the QT side pointer, its reset helper # and the expert-slot wiring are exercised as the engine actually builds them. tests/test_xdna_qt_state$(EXE): tests/test_xdna_qt_state.c colibri.c backend_xdna.c backend_xdna.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) -DCOLI_XDNA $< backend_xdna.c -o $@ $(LDFLAGS) -tests/test_xdna_prepared_state$(EXE): tests/test_xdna_prepared_state.c backend_xdna.c backend_xdna.h compat.h +tests/test_xdna_prepared_state$(EXE): tests/test_xdna_prepared_state.c backend_xdna.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_xdna_registry$(EXE): tests/test_xdna_registry.c backend_xdna.c backend_xdna.h +tests/test_xdna_registry$(EXE): tests/test_xdna_registry.c backend_xdna.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Synthetic ABI-2 helpers for the execution tests. Plain C, no XRT, no device: @@ -1375,7 +1390,7 @@ tests/xdna_fake_helper_abi1.dll: tests/xdna_fake_helper.c tests/xdna_fake_helper_partial.dll: tests/xdna_fake_helper.c $(CC) $(CFLAGS) -DFAKE_PARTIAL -shared $< -o $@ -tests/test_xdna_execution$(EXE): tests/test_xdna_execution.c backend_xdna.c backend_xdna.h compat.h $(XDNA_FAKE_HELPERS) +tests/test_xdna_execution$(EXE): tests/test_xdna_execution.c backend_xdna.c $(XDNA_FAKE_HELPERS) $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Failure/fallback owner. Needs BOTH the engine (matmul_qt is the fallback @@ -1396,45 +1411,45 @@ tests/test_xdna_failure$(EXE): tests/test_xdna_failure.c colibri.c backend_xdna. tests/xdna_physical_probe$(EXE): tests/xdna_physical_probe.c colibri.c backend_xdna.c backend_xdna.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) -DCOLI_XDNA $< backend_xdna.c -o $@ $(LDFLAGS) -tests/test_json$(EXE): tests/test_json.c json.h +tests/test_json$(EXE): tests/test_json.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_tok_o200k$(EXE): tests/test_tok_o200k.c tok.h tok_unicode.h tok_unicode_o200k.h json.h +tests/test_tok_o200k$(EXE): tests/test_tok_o200k.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_tok_gpt2$(EXE): tests/test_tok_gpt2.c tok.h tok_unicode.h tok_unicode_o200k.h json.h +tests/test_tok_gpt2$(EXE): tests/test_tok_gpt2.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ) +tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c kimi_k3.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ) +tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) -tests/test_tok_kimi_tiny$(EXE): tests/test_tok_kimi_tiny.c tok.h tok_unicode.h tok_unicode_o200k.h json.h +tests/test_tok_kimi_tiny$(EXE): tests/test_tok_kimi_tiny.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_pread$(EXE): tests/test_st_pread.c st.h json.h compat.h +tests/test_st_pread$(EXE): tests/test_st_pread.c $(CC) $(CFLAGS) -DST_PREAD_CHUNK=7 $< -o $@ $(LDFLAGS) -tests/test_st_slice$(EXE): tests/test_st_slice.c st.h json.h compat.h +tests/test_st_slice$(EXE): tests/test_st_slice.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< edge_runtime.c -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_dsv41_serve_budget$(EXE): tests/test_dsv41_serve_budget.c sse41_kernels.h deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h +tests/test_dsv41_serve_budget$(EXE): tests/test_dsv41_serve_budget.c deepseek_v41.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_idot$(EXE): tests/test_qwen38_idot.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_idot$(EXE): tests/test_qwen38_idot.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_vision$(EXE): tests/test_qwen38_vision.c qwen38_vision.h st.h json.h compat.h +tests/test_qwen38_vision$(EXE): tests/test_qwen38_vision.c $(CC) $(CFLAGS) -Wno-unused-function $< -o $@ $(LDFLAGS) # Genera la fixture della torre e la confronta con l'oracolo upstream. Non serve @@ -1453,33 +1468,33 @@ qwen38-vision-serve-check: qwen38$(EXE) $(PYTHON) tools/make_edge_tiny_tokenizer.py --vocab-size 64 ./qwen38_mm_tiny $(PYTHON) tests/test_qwen38_vision_serve.py --binary ./qwen38$(EXE) --fixture ./qwen38_mm_tiny -tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< segment_runtime.c -o $@ $(NOCUDA_LDFLAGS) -tests/test_st_map$(EXE): tests/test_st_map.c st.h json.h compat.h +tests/test_st_map$(EXE): tests/test_st_map.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_dup_name_refusal$(EXE): tests/test_dup_name_refusal.c st.h json.h compat.h +tests/test_dup_name_refusal$(EXE): tests/test_dup_name_refusal.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_shape$(EXE): tests/test_st_shape.c st.h json.h compat.h +tests/test_st_shape$(EXE): tests/test_st_shape.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Exhaustive bit-exact gate for st.h's bf16_to_f32_bulk/f16_to_f32_bulk AVX2 tier # (all 65536 patterns per format -- see the file for why no tolerance applies). -tests/test_st_f16_bf16_simd$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h +tests/test_st_f16_bf16_simd$(EXE): tests/test_st_f16_bf16_simd.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Same gate, forced onto the SSE4.1 tier: overrides the host's default -march # so the SSE4.1 body in st.h is actually exercised even on a devbox that # would otherwise always pick the AVX2 tier. -tests/test_st_f16_bf16_simd_sse41$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h +tests/test_st_f16_bf16_simd_sse41$(EXE): tests/test_st_f16_bf16_simd.c $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) # bf16_to_f32_bulk/f16_to_f32_bulk vs the scalar per-element reference, NOT a @@ -1493,28 +1508,28 @@ tests/bench_st_f16_bf16_simd$(EXE): tests/bench_st_f16_bf16_simd.c st.h json.h c tests/bench_load_tq_peak_rss$(EXE): tests/bench_load_tq_peak_rss.c $(CC) -O3 -march=native $< -o $@ -lm -tests/test_st$(EXE): tests/test_st.c st.h json.h compat.h +tests/test_st$(EXE): tests/test_st.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_mirror$(EXE): tests/test_st_mirror.c st.h json.h compat.h +tests/test_st_mirror$(EXE): tests/test_st_mirror.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_overlay$(EXE): tests/test_st_overlay.c st.h json.h compat.h +tests/test_st_overlay$(EXE): tests/test_st_overlay.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_compat_mem$(EXE): tests/test_compat_mem.c compat.h +tests/test_compat_mem$(EXE): tests/test_compat_mem.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_missing$(EXE): tests/test_st_missing.c st.h json.h compat.h +tests/test_st_missing$(EXE): tests/test_st_missing.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_st_range_rep$(EXE): tests/test_st_range_rep.c st.h json.h compat.h +tests/test_st_range_rep$(EXE): tests/test_st_range_rep.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_tier$(EXE): tests/test_tier.c tier.h +tests/test_tier$(EXE): tests/test_tier.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_grammar$(EXE): tests/test_grammar.c grammar.h +tests/test_grammar$(EXE): tests/test_grammar.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # RoPE inv_freq precompute must stay byte-identical to the old powf-per-position @@ -1527,12 +1542,12 @@ tests/test_grammar$(EXE): tests/test_grammar.c grammar.h # equivalence under the shipping flags is covered by the differential dump # (old vs new rope_interleave are byte-identical); this test guards the # source-level formula equivalence and must not depend on contraction luck. -tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) -ffp-contract=off $< $(VK_OBJ) -o $@ $(LDFLAGS) # schema->GBNF compile cache (#7): grammar_reset must equal a fresh setup, and # the GrDraft.src ownership must not leak/double-free. Includes colibri.c. -tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # The classified logit-row reduction in sample.h and the digest in @@ -1540,18 +1555,18 @@ tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c sse41_kernels.h colib # agreement check against the plain logprob_target the sampling path uses # on rows a float subtraction cannot mis-round, and a double-precision # check against a second double computation on rows that do. -tests/test_logprob_status$(EXE): tests/test_logprob_status.c colibri.c sse41_kernels.h exact_dot.h oracle.h sample.h evidence_digest.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h kv_persist.h telemetry.h route_trace.h +tests/test_logprob_status$(EXE): tests/test_logprob_status.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # The ablation mode's manifest loader, evidence writer and dispatch contract. # The adapter build bypasses only the model computation, so the parser/writer # path is exercised for real with no model and no weights. -tests/test_ablate_mode$(EXE): tests/test_ablate_mode.c colibri.c sse41_kernels.h exact_dot.h oracle.h sample.h evidence_digest.h abl.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h kv_persist.h telemetry.h route_trace.h +tests/test_ablate_mode$(EXE): tests/test_ablate_mode.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Standalone: drives a faithful miniature of moe()'s routing+accumulate and links # the SAME abl.h the engine links -- no model/weights needed (the ablation-logic gate). -tests/test_ablate$(EXE): tests/test_ablate.c abl.h +tests/test_ablate$(EXE): tests/test_ablate.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Standalone: exercises the DEGRADE_ZERO miss-slot zero-fill logic extracted from @@ -1559,22 +1574,22 @@ tests/test_ablate$(EXE): tests/test_ablate.c abl.h tests/test_degrade_zero$(EXE): tests/test_degrade_zero.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h json.h +tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h +tests/test_decode_batch$(EXE): tests/test_decode_batch.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_pin_pool$(EXE): tests/test_pin_pool.c pin_pool.h kv_prefix.h +tests/test_pin_pool$(EXE): tests/test_pin_pool.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_serve_codec$(EXE): tests/test_serve_codec.c serve_codec.h compat.h +tests/test_serve_codec$(EXE): tests/test_serve_codec.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_serve_budget$(EXE): tests/test_serve_budget.c serve_budget.h +tests/test_serve_budget$(EXE): tests/test_serve_budget.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c segment_runtime.c segment_runtime.h @@ -1583,10 +1598,10 @@ tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c segment_runtime.c tests/test_segment_conformance$(EXE): tests/test_segment_conformance.c tests/segment_conformance_fixtures.c tests/segment_conformance_fixtures.h segment_runtime.c segment_runtime.h $(CC) $(CFLAGS) tests/test_segment_conformance.c tests/segment_conformance_fixtures.c segment_runtime.c -o $@ $(LDFLAGS) -tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h serve_budget.h $(INK_CUDA_OBJ) $(METAL_OBJ) +tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) -tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) +tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) # Reproducible local A/B for the shared-expert prefill path. This is timing @@ -1594,56 +1609,56 @@ tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling tests/bench_inkling_shared_batch$(EXE): tests/bench_inkling_shared_batch.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) -tests/test_inkling_cache_index$(EXE): tests/test_inkling_cache_index.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) +tests/test_inkling_cache_index$(EXE): tests/test_inkling_cache_index.c inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) -tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c sse41_kernels.h serve_budget.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) +tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) -tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) +tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) -tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) +tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) # olmoe's matmul_q, not colibri's: compares the IDOT path against FP32 # ACTIVATIONS rather than an integer reference. NOCUDA_* for the same reason # the olmoe target uses it -- olmoe.c contains no COLI_CUDA code. -tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c olmoe.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c serve_budget.h sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c olmoe.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c olmoe.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) - $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) +tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -tests/test_idot$(EXE): tests/test_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_idot$(EXE): tests/test_idot.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) -DCOLI_HAVE_GROUPED_PAIR $< $(VK_OBJ) -o $@ $(LDFLAGS) # Force the grouped-int4 oracle onto the SSE4.1 tier even on an AVX2 host. -tests/test_i4_grouped_sse41$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_i4_grouped_sse41$(EXE): tests/test_i4_grouped.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) -O3 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_i4_grouped_sse41_o1$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_i4_grouped_sse41_o1$(EXE): tests/test_i4_grouped.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) -O1 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_i4_grouped_sse41_no_contract$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_i4_grouped_sse41_no_contract$(EXE): tests/test_i4_grouped.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) -O3 -ffp-contract=off -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_stops$(EXE): tests/test_stops.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_stops$(EXE): tests/test_stops.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_topp$(EXE): tests/test_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_topp$(EXE): tests/test_topp.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test @@ -1651,42 +1666,42 @@ tests/test_topp$(EXE): tests/test_topp.c sse41_kernels.h colibri.c oracle.h st.h tests/bench_topp$(EXE): tests/bench_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_sample_nan$(EXE): tests/test_sample_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_temp_env$(EXE): tests/test_temp_env.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_temp_env$(EXE): tests/test_temp_env.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ) +tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c kv_fp8.h decode_batch.h +tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_kv_tq$(EXE): tests/test_kv_tq.c kv_tq.h +tests/test_kv_tq$(EXE): tests/test_kv_tq.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_kv_disk$(EXE): tests/test_kv_disk.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ) +tests/test_kv_disk$(EXE): tests/test_kv_disk.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # fmt=6 kernel oracle: needs the generated grid table, and its fixture comes from # the reference codec (tools/make_e8_fixture.py) — regenerate if the layout moves. -tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c sse41_kernels.h quant.h fp8_format.h idot.h +tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c sse41_kernels.h quant.h fp8_format.h idot.h +tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_stop_ids$(EXE): tests/test_stop_ids.c stop_ids.h json.h +tests/test_stop_ids$(EXE): tests/test_stop_ids.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_sparse_attn$(EXE): tests/test_sparse_attn.c sparse_attn.h +tests/test_sparse_attn$(EXE): tests/test_sparse_attn.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_expert_ffn$(EXE): tests/test_expert_ffn.c expert_ffn.h +tests/test_expert_ffn$(EXE): tests/test_expert_ffn.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_rans$(EXE): tests/test_rans.c rans.h compat.h +tests/test_rans$(EXE): tests/test_rans.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Structured-mutation fuzz for the rans record parser/decoder under @@ -1701,47 +1716,47 @@ fuzz-rans: tests/fuzz_rans.c rans.h tests/fuzz_rans.c -o tests/fuzz_rans -lm ./tests/fuzz_rans -tests/test_int3$(EXE): tests/test_int3.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_int3$(EXE): tests/test_int3.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_int3_load$(EXE): tests/test_int3_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_int3_load$(EXE): tests/test_int3_load.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c sse41_kernels.h quant.h fp8_format.h idot.h +tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Host-only: backend_cuda.h's format predicate is plain C, so its truth table is # a portable gate even where there is no CUDA toolchain. Same arrangement as # metal_fused_fmt_ok's CPU-testable truth table. -tests/test_cuda_fmt_guard$(EXE): tests/test_cuda_fmt_guard.c backend_cuda.h +tests/test_cuda_fmt_guard$(EXE): tests/test_cuda_fmt_guard.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # The fmt=8 LUT-gate state machine. Like test_cuda_fmt_guard it links no CUDA # object and needs no toolchain: the two decisions live as pure predicates in # backend_cuda.h and backend_cuda.cu calls them, so this pins the engine's own # logic on a plain CPU build. -tests/test_cuda_lut_gate$(EXE): tests/test_cuda_lut_gate.c backend_cuda.h +tests/test_cuda_lut_gate$(EXE): tests/test_cuda_lut_gate.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_fp8_load$(EXE): tests/test_fp8_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_fp8_load$(EXE): tests/test_fp8_load.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_logit_nan$(EXE): tests/test_logit_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) +tests/test_logit_nan$(EXE): tests/test_logit_nan.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_router_nan$(EXE): tests/test_router_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) +tests/test_router_nan$(EXE): tests/test_router_nan.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h +tests/test_compat_direct$(EXE): tests/test_compat_direct.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_expert_store_ops$(EXE): tests/test_expert_store_ops.c expert_store.h tensor.h +tests/test_expert_store_ops$(EXE): tests/test_expert_store_ops.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_native_quant$(EXE): tests/test_native_quant.c sse41_kernels.h deepseek_v4.c native_quant.h tensor.h quant.h fp8_format.h idot.h @@ -1999,40 +2014,40 @@ tests/test_v4_serve_framing$(EXE): tests/test_v4_serve_framing.c sse41_kernels.h endif -tests/test_route_trace$(EXE): tests/test_route_trace.c route_trace.h compat.h +tests/test_route_trace$(EXE): tests/test_route_trace.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_cli_args$(EXE): tests/test_cli_args.c cli_args.h +tests/test_cli_args$(EXE): tests/test_cli_args.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_oracle$(EXE): tests/test_oracle.c oracle.h json.h +tests/test_oracle$(EXE): tests/test_oracle.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Il tier CUDA con un backend finto: gira SENZA GPU perche' il test definisce i # coli_cuda_* e registra cosa riceve. E' il solo modo di provare in CI che un # esperto arriva davvero in VRAM e nel formato giusto (#1331) -- un test che si # fermasse a "qt_init ritorna 1" sarebbe passato anche col difetto. -tests/test_qwen36_tier_int8$(EXE): tests/test_qwen36_tier_int8.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_int8$(EXE): tests/test_qwen36_tier_int8.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend (tests/qwen36_fake_cuda.h), two devices: proves the tier's # per-device input-replica block (G.is_x) is sized and strided so device 1's # block cannot run past the end of the buffer or into device 0's (#1339). -tests/test_qwen36_tier_multidev$(EXE): tests/test_qwen36_tier_multidev.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_multidev$(EXE): tests/test_qwen36_tier_multidev.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Budget separate gate/up/down scale allocations at their own size classes. -tests/test_qwen36_tier_scale_budget$(EXE): tests/test_qwen36_tier_scale_budget.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_scale_budget$(EXE): tests/test_qwen36_tier_scale_budget.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend, single device: proves qt_shutdown returns (under an # alarm(10) watchdog) instead of hanging forever when a group is still open # and an LFRU swap is parked waiting for cv_take (#1340). -tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Release owned projections and experts before CUDA teardown, then reopen safely. -tests/test_qwen36_tier_release$(EXE): tests/test_qwen36_tier_release.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_release$(EXE): tests/test_qwen36_tier_release.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend, but driving the ENGINE: the warmstart in qwen36.c hands the @@ -2047,18 +2062,18 @@ tests/test_qwen36_tier_release$(EXE): tests/test_qwen36_tier_release.c tests/qwe # issue blocks stay inside, disjoint and at their device's slot under random # routing on 1-3 devices. Under test-asan the third doubles as a fuzz for the # #1339 class. -tests/test_qwen36_tier_invariants$(EXE): tests/test_qwen36_tier_invariants.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_invariants$(EXE): tests/test_qwen36_tier_invariants.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Async collection errors must drain all devices and reject partial output. -tests/test_qwen36_tier_take_error$(EXE): tests/test_qwen36_tier_take_error.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_take_error$(EXE): tests/test_qwen36_tier_take_error.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend, with uploads that take time: qt_fill_wait must not return # until the last enqueued expert is RESIDENT, not merely dequeued -- the engine # frees the RAM int8 copies right after it (#1360 saw the gap one run in # fifteen; here the slow upload hook makes it every run without the fix). -tests/test_qwen36_tier_fill_wait$(EXE): tests/test_qwen36_tier_fill_wait.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_fill_wait$(EXE): tests/test_qwen36_tier_fill_wait.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend: the automatic placement (COLI_PLACE unset/"auto") puts @@ -2066,54 +2081,54 @@ tests/test_qwen36_tier_fill_wait$(EXE): tests/test_qwen36_tier_fill_wait.c tests # experts by heat, keeps the trunk on the CPU when the marginal experts are # hotter, honours an explicit list and "off", and takes placed bytes out of # the expert budget -- on one and on two devices. -tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Failed tier startup must release host storage and initialized synchronization. -tests/test_qwen36_tier_init_failure$(EXE): tests/test_qwen36_tier_init_failure.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_init_failure$(EXE): tests/test_qwen36_tier_init_failure.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Same fake backend: the fp8 streaming mode a model whose experts do not fit # in RAM needs (Qwen3.8) -- cap < n_experts accepted, fmt=8 uploads with # 128x128 block scales, bytes staged unchanged, no pointer kept into the # engine's recycled slot, promotion decided when the bytes pass by. -tests/test_qwen36_tier_fp8$(EXE): tests/test_qwen36_tier_fp8.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_fp8$(EXE): tests/test_qwen36_tier_fp8.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Failed gate/up/down uploads must release every unpublished CUDA tensor. -tests/test_qwen36_tier_rollback$(EXE): tests/test_qwen36_tier_rollback.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h +tests/test_qwen36_tier_rollback$(EXE): tests/test_qwen36_tier_rollback.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # Generic resident dense matrices (the Qwen3.8 trunk) and per-offer placement. -tests/test_qwen36_tier_dense$(EXE): tests/test_qwen36_tier_dense.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h +tests/test_qwen36_tier_dense$(EXE): tests/test_qwen36_tier_dense.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # DeltaNet input projection batching, state parity and block failure fallback. -tests/test_qwen36_dnproj_batch$(EXE): tests/test_qwen36_dnproj_batch.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h quant.h +tests/test_qwen36_dnproj_batch$(EXE): tests/test_qwen36_dnproj_batch.c qwen36.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # the automatic trunk placement can be withdrawn after the startup probe (fake backend). -tests/test_qwen36_tier_withdraw$(EXE): tests/test_qwen36_tier_withdraw.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h +tests/test_qwen36_tier_withdraw$(EXE): tests/test_qwen36_tier_withdraw.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) -tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h +tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c qwen36.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # dnout / attnproj / shexp offered, placed and served from VRAM (fake backend). -tests/test_qwen36_trunk_dense$(EXE): tests/test_qwen36_trunk_dense.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h +tests/test_qwen36_trunk_dense$(EXE): tests/test_qwen36_trunk_dense.c qwen36.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # #1391: il decode path deve offrire gli esperti int8 al tier, non solo il warmstart -tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h +tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c qwen36.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # The qwen38 engine through its own main() on the fake backend: tier start # from the FP8 fixture, reduced CPU list, qt_note on recycled slots, oracle # tokens unchanged. Needs the fixture (qwen38-tiny-fp8-generate). -tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c sse41_kernels.h tests/qwen36_fake_cuda.h qwen38.c qwen38_core.h kv_prefix.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h +tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c qwen38.c qwen36_tier.c $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) -tests/test_serve_poll$(EXE): tests/test_serve_poll.c serve_poll.h +tests/test_serve_poll$(EXE): tests/test_serve_poll.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Linux-only: legge /proc/self/status. La regola c'e' comunque, il test si @@ -2123,47 +2138,47 @@ tests/test_rss_anon$(EXE): tests/test_rss_anon.c # #1375: la memoria disponibile e' misurata in compat.h per Linux, macOS e # Windows; questo test gira nei tre job e fallisce dove la misura vale 0. -tests/test_mem_available$(EXE): tests/test_mem_available.c compat.h +tests/test_mem_available$(EXE): tests/test_mem_available.c $(CC) $(CFLAGS) tests/test_mem_available.c -o $@ $(LDFLAGS) -tests/test_exact_dot$(EXE): tests/test_exact_dot.c exact_dot.h +tests/test_exact_dot$(EXE): tests/test_exact_dot.c $(CC) $(CFLAGS) tests/test_exact_dot.c -o tests/test_exact_dot$(EXE) $(LDFLAGS) -tests/test_798_guards$(EXE): tests/test_798_guards.c st.h json.h compat.h route_trace.h +tests/test_798_guards$(EXE): tests/test_798_guards.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_dsa_select$(EXE): tests/test_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_dsa_select$(EXE): tests/test_dsa_select.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_v4_hybrid_policy$(EXE): tests/test_v4_hybrid_policy.c deepseek_v4_hybrid.h hybrid_split.h +tests/test_v4_hybrid_policy$(EXE): tests/test_v4_hybrid_policy.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_k3_fill_budget$(EXE): tests/test_k3_fill_budget.c hybrid_split.h +tests/test_k3_fill_budget$(EXE): tests/test_k3_fill_budget.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_v4_bank_pair$(EXE): tests/test_v4_bank_pair.c deepseek_v4_bank_pair.h +tests/test_v4_bank_pair$(EXE): tests/test_v4_bank_pair.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356), @@ -2202,67 +2217,67 @@ tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c sse41_kernels.h colibri tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_uring$(EXE): tests/test_uring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) +tests/test_uring$(EXE): tests/test_uring.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_pipe_block$(EXE): tests/test_pipe_block.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) +tests/test_pipe_block$(EXE): tests/test_pipe_block.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_omp_tune$(EXE): tests/test_omp_tune.c omp_tune.h compat.h +tests/test_omp_tune$(EXE): tests/test_omp_tune.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_compat_env$(EXE): tests/test_compat_env.c compat.h +tests/test_compat_env$(EXE): tests/test_compat_env.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) +tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c colibri.c $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # Standalone: proves the group-scaled int8 GEMV keeps every row's float # operations in their original order -- the assumption qwen36's byte-identical # output rests on. Links the SAME gsgemv.h the engine links. -tests/test_gsgemv$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h +tests/test_gsgemv$(EXE): tests/test_gsgemv.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Same gate, forced onto the SSE4.1 tier: overrides the host's default -march # so the SSE4.1 body in gsgemv.h is actually exercised even on a devbox that # would otherwise always pick the AVX2 tier. -tests/test_gsgemv_sse41$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h +tests/test_gsgemv_sse41$(EXE): tests/test_gsgemv.c $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) # Standalone: proves the plain (non-group-scaled) int8 GEMV keeps its exact # sequence of float operations -- the assumption qwen36's byte-identical # output rests on. Links the SAME qgemv.h the engine links. -tests/test_qgemv$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h +tests/test_qgemv$(EXE): tests/test_qgemv.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Same gate, forced onto the SSE4.1 tier: overrides the host's default -march # so the SSE4.1 body in qgemv.h is actually exercised even on a devbox that # would otherwise always pick the AVX2 tier. -tests/test_qgemv_sse41$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h +tests/test_qgemv_sse41$(EXE): tests/test_qgemv.c $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) # olmoe's dot_i8_16 (ARM NEON/AVX2/SSE4.1 variants): must be bit-exact against # a scalar int8 reference -- pure integer arithmetic, no tolerance needed. # NOCUDA_* for the same reason the other olmoe.c-including test rules use it. -tests/test_olmoe_dot_i8_16$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_dot_i8_16$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) # Same gate, forced onto the SSE4.1 tier: overrides the host's default -march # so the SSE4.1 body in olmoe.c's dot_i8_16 is actually exercised even on a # devbox that would otherwise always pick the AVX2 tier. -tests/test_olmoe_dot_i8_16_sse41$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_dot_i8_16_sse41$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c $(CC) $(NOCUDA_CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(NOCUDA_LDFLAGS) # layer_cuda_shard_kvb's format-allowlist refusal (see the test's file header). The # function only exists under -DCOLI_CUDA, so on a default (CPU) build the binary is a # loud SKIP; built with CUDA=1 it links backend_cuda.o ($(CUDA_OBJ)) and exercises the # real guard -- no GPU needed, every probed path returns before any device context. -tests/test_shard_kvb_refuse$(EXE): tests/test_shard_kvb_refuse.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h sample.h kv_persist.h telemetry.h $(CUDA_OBJ) $(VK_OBJ) +tests/test_shard_kvb_refuse$(EXE): tests/test_shard_kvb_refuse.c colibri.c $(CUDA_OBJ) $(VK_OBJ) $(CC) $(CFLAGS) $< $(VK_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS) # The e8x4g64 container loader harness. It takes a minted directory on argv and @@ -2270,7 +2285,7 @@ tests/test_shard_kvb_refuse$(EXE): tests/test_shard_kvb_refuse.c colibri.c st.h # exists so it also compiles under the suite's own $(CFLAGS) rather than only # under the driver's hand-copied flag list, and so a warning regression in it # fails the normal build. -tests/test_e8x4g64_loader$(EXE): tests/test_e8x4g64_loader.c st.h quant.h fp8_format.h compat.h +tests/test_e8x4g64_loader$(EXE): tests/test_e8x4g64_loader.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Reachable build-only entry point for the harness above: `make check` never @@ -2412,12 +2427,50 @@ bench: iobench$(EXE) .PHONY: test-asan .PHONY: dsv4-cuda-test cuda-dsv4-dll cuda-dsv4-dg-dll dsv4-cuda-loader-test -tests/test_kv_prefix$(EXE): tests/test_kv_prefix.c kv_prefix.h +tests/test_kv_prefix$(EXE): tests/test_kv_prefix.c $(CC) $(CFLAGS) tests/test_kv_prefix.c -o tests/test_kv_prefix$(EXE) $(LDFLAGS) -tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) +tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) tests/test_kimi_request_state.c $(VK_OBJ) -o tests/test_kimi_request_state$(EXE) $(NOCUDA_LDFLAGS) # Kimi CUDA dispatch/fallback with a fake backend; no CUDA toolkit required. -tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c sse41_kernels.h kimi_k3.c backend_cuda.h kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) +tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) -DCOLI_CUDA $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) + +# --------------------------------------------------------------------------- +# Generated header prerequisites (#1741) +# +# Every rule in AUTODEP_BINS compiles ONE translation unit with $(CFLAGS), so +# -MMD (added to CFLAGS near the top) leaves a complete .d and the rule +# does not list headers by hand. The engines and tests are derived the same way +# TEST_RULES is, so a new rule joins automatically. +# +# Two kinds of rule keep hand-written header lists instead: +# - AUTODEP_MULTI_TU compiles more than one .c in a single command. GCC then +# writes only the LAST unit's dependencies, so a .d would silently miss the +# headers of the others. tests/test_makefile_deps.py fails if a recipe of +# that shape appears anywhere in AUTODEP_BINS. +# - AUTODEP_ELSEWHERE is built outside this Makefile's own $(CC) recipes +# (Makefile.deepseek-v4 and the adapter test targets), as are the nvcc, +# MSVC-hosted CUDA and Metal objects, which are not listed here at all. +# +# Most rules do not depend on .build-config, so a change to CFLAGS alone would +# not rebuild a binary made before this change, and it would have no .d. Each +# target therefore also depends on its own .d through an empty rule: a missing +# .d counts as out of date, the target is rebuilt, and that compile writes it. +# The same covers a .d deleted by hand. +AUTODEP_MULTI_TU = test_xdna_qt_state test_xdna_failure test_qwen38_tokenizer \ + test_qwen38_native_weights test_segment_runtime test_segment_conformance \ + test_native_quant test_edge_runtime +AUTODEP_ELSEWHERE = test_segment_adapters_registration test_segment_adapters_real \ + test_edge_adapters_registration test_edge_adapters_real test_deepseek_v4 \ + test_v4_serve_framing +ENGINE_RULES := $(foreach n,$(shell sed -n 's|^\([a-z0-9_]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST))),$(if $(wildcard $(n).c),$(n))) +AUTODEP_OBJS = backend_vulkan.o backend_loader.o backend_xdna.o qwen36_tier.o +AUTODEP_BINS = $(addsuffix $(EXE),$(ENGINE_RULES)) \ + $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_MULTI_TU) $(AUTODEP_ELSEWHERE),$(TEST_RULES)))) \ + $(AUTODEP_OBJS) +AUTODEP_DEPS = $(addsuffix .d,$(basename $(AUTODEP_BINS))) +$(foreach b,$(AUTODEP_BINS),$(eval $(b): $(basename $(b)).d)) +$(AUTODEP_DEPS): ; +-include $(wildcard $(AUTODEP_DEPS)) diff --git a/c/tests/test_glm53_metal_source.py b/c/tests/test_glm53_metal_source.py index f9ea6ceae..bddc1b511 100644 --- a/c/tests/test_glm53_metal_source.py +++ b/c/tests/test_glm53_metal_source.py @@ -40,7 +40,8 @@ def test_glm53_links_metal_object(self): start = MAKE.index("glm53$(EXE):") end = MAKE.index("kimi_k3$(EXE):", start) rule = MAKE[start:end] - self.assertIn("backend_metal.h", rule) + # backend_metal.h is no longer listed by hand: glm53.c includes it + # under COLI_METAL, so a METAL=1 build's glm53.d names it (#1741). self.assertIn("$(METAL_OBJ)", rule) diff --git a/c/tests/test_makefile_deps.py b/c/tests/test_makefile_deps.py index cfb2903a7..8462533da 100644 --- a/c/tests/test_makefile_deps.py +++ b/c/tests/test_makefile_deps.py @@ -1,76 +1,192 @@ -"""Makefile header prerequisites must cover what each engine actually includes. +"""Header prerequisites are generated by the compiler, and must stay that way. -`make` tracks file timestamps, not #include graphs, and this Makefile has no --MMD/-include dependency generation. So a header that an engine includes but -that is absent from its rule's prerequisite list produces the worst kind of +`make` tracks file timestamps, not #include graphs. A header an engine includes +but that is absent from its rule's prerequisites produces the worst kind of build result: `make` reports success, changes nothing, and leaves a STALE -binary that looks freshly built. - -Reproduce on a tree without the accompanying fix, without a compiler: - - touch colibri && sleep 1 && touch edge_runtime.h - make -q colibri; echo $? - -> 0 # "up to date", though colibri.c includes edge_runtime.h - -whereas touching a header the rule DOES list (st.h) exits 1, meaning make -would relink. This test is the enforcement a comment cannot provide: nothing -else compares the two lists, and the drift is invisible from inside a green -build. - -The engine list is DERIVED from the Makefile, never kept here. A hand-written -list can only catch a rule that was renamed or removed; it cannot catch a rule -that was never added, which is the case that actually happened -- glm53 and -qwen38 sat with incomplete prerequisites while a hand-listed version of this -test passed green. Anything matching `NAME$(EXE):` with a `NAME.c` beside it -is a target built from one translation unit, so its prerequisites are -checkable, and the next engine added is covered without anyone remembering. +binary that looks freshly built (#1284). + +#1284 closed that by listing every header by hand and checking the lists here. +The lists then became the most edited line in the Makefile -- 27 commits by 16 +authors in 30 days, and a merge conflict for every open PR touching the same +rule (#1741). So the lists are gone: `-MMD -MP` in CFLAGS makes the compiler +write `.d` naming each header that compile actually read, and the +Makefile includes those files. What this file checks is that the generated +dependencies really are complete, for three ways they could silently not be: + +1. A recipe that compiles more than one .c in one command. GCC then writes + only the LAST unit's dependencies, so the others' headers are missing from + the .d with nothing to show for it. Such recipes keep hand-written lists + (AUTODEP_MULTI_TU) and every rule in AUTODEP_BINS is checked to compile + exactly one unit, with -MMD, by asking make for the commands it would run. + +2. A binary built without a .d, e.g. before this change. Most rules do not + depend on .build-config, so nothing else would force the rebuild that writes + one. The Makefile makes each target depend on its own .d for that reason; + the coverage test below fails on any built target whose .d is missing. + +3. A .d that does not cover what its source includes. Every unconditional + `#include "..."` of each built target's source must appear in its .d, and + for the engines `make -q -W
` must actually answer "rebuild". + Includes under #if are not required: `uring.h` is read on Linux only, and a + Windows build that does not depend on it is correct. + +The engine list is still DERIVED from the Makefile, never kept here: anything +matching `NAME$(EXE):` with a `NAME.c` beside it is an engine, and the family +registry is the cross-check that the parse found them all. """ +import os import re +import shlex +import shutil +import subprocess import unittest from pathlib import Path from family_registry import FAMILIES C_DIR = Path(__file__).resolve().parent.parent +MAKE = shutil.which("make") -# `NAME$(EXE): prereqs`. Targets with a path separator (tests/foo$(EXE)) are -# deliberately excluded: this checks the engines built at the top of c/. RULE_RE = re.compile(r"(?m)^([A-Za-z0-9_]+)\$\(EXE\):[ \t]*(.*)$") -INCLUDE_RE = re.compile(r'^[ \t]*#[ \t]*include[ \t]*"([^"]+)"', re.M) +INCLUDE_RE = re.compile(r'^[ \t]*#[ \t]*include[ \t]*"([^"]+)"') +COND_OPEN_RE = re.compile(r"^[ \t]*#[ \t]*if") +COND_CLOSE_RE = re.compile(r"^[ \t]*#[ \t]*endif") +SOURCE_RE = re.compile(r"\.(c|cc|cpp|m)$") # deepseek_v4 is built by Makefile.deepseek-v4 through a phony delegating -# target, so it has no `NAME$(EXE):` rule here and no prerequisite list of its -# own to check. test_family_registry pins that separately. +# target, so it has no `NAME$(EXE):` rule here. test_family_registry pins that. SEPARATE_MAKEFILE = {"deepseek-v4"} -def _engine_rules(): - """Every target built from a single same-named .c, taken from the Makefile. - - Line continuations are folded first: a rule wrapped across lines would - otherwise present a truncated prerequisite list and report headers as - missing that are listed on the next line. - """ +def _joined_makefile(): text = (C_DIR / "Makefile").read_text(encoding="utf-8") - joined = re.sub(r"\\\n[ \t]*", " ", text) + return re.sub(r"\\\r?\n[ \t]*", " ", text).replace("\r", "") + + +def _engine_rules(): + """Every target built from a single same-named .c, taken from the Makefile.""" rules = {} - for name, prereqs in RULE_RE.findall(joined): - source = C_DIR / f"{name}.c" - if source.exists(): - rules[name] = (source, set(prereqs.split("#", 1)[0].split())) + for name, prereqs in RULE_RE.findall(_joined_makefile()): + if (C_DIR / f"{name}.c").exists(): + rules[name] = prereqs.split("#", 1)[0].split() return rules -class MakefileHeaderDepsTest(unittest.TestCase): +def _make(*args): + """Run make in c/ with the build's own configuration. + + Inside `make check` the child inherits MAKEFLAGS, so it parses the same + variables the build did and leaves .build-config as it was. + """ + return subprocess.run([MAKE, "--no-print-directory", *args], cwd=C_DIR, + text=True, encoding="utf-8", errors="replace", + capture_output=True, timeout=600) + + +def _make_var(name): + # `make --eval` needs GNU Make 3.82; macOS ships 3.81, which does read an + # extra makefile from stdin. + proc = subprocess.run([MAKE, "-s", "--no-print-directory", "-f", "Makefile", + "-f", "-", f"print-{name}"], cwd=C_DIR, text=True, + input="print-%: ; @echo $($*)\n", capture_output=True, + timeout=300) + if proc.returncode != 0: + raise AssertionError(f"could not read {name} from the Makefile:\n" + f"{proc.stderr}") + return proc.stdout.split() + + +def _commands(targets, *variables): + """The compile command make WOULD run for each target, keyed by its -o. + + -B as well as -n: an already-built target otherwise prints no recipe. + """ + proc = _make("-Bn", *targets, *variables) + if proc.returncode != 0: + raise AssertionError(f"make -Bn {' '.join(variables)} failed:\n" + f"{proc.stderr[-2000:]}") + commands = {} + for line in re.sub(r"\\\r?\n", " ", proc.stdout).splitlines(): + try: + words = shlex.split(line) + except ValueError: + continue + if "-o" in words and words.index("-o") + 1 < len(words): + commands[words[words.index("-o") + 1]] = words + return commands + + +def _gpu_variables(exe): + """The make variable that builds the GPU flavour here, or None. + + QWEN36_TIER_OBJ only exists with CUDA/HIP, and a GPU build is where a + second unit last crept onto a command line, so the one-unit check has to + see that configuration too. CUDA=1 is refused off Linux and the wording + differs per platform, so test the fact rather than the message, the way + test_makefile_cuda_scope does: does colibri then get -DCOLI_CUDA? + """ + for variable in ("CUDA=1", "CUDA_DLL=1"): + proc = _make("-Bn", "colibri" + exe, variable) + if proc.returncode == 0 and "-DCOLI_CUDA" in proc.stdout: + return variable + return None + + +def _dep_file(target): + return C_DIR / (os.path.splitext(target)[0] + ".d") + + +def _raw_deps(dep_file): + """The prerequisites of the first rule in a .d file, spelled as make sees + them: a test's include of "../st.h" is `tests/../st.h`, not `st.h`.""" + text = dep_file.read_text(encoding="utf-8", errors="replace") + text = re.sub(r"\\\r?\n", " ", text) + first = text.split("\n", 1)[0] + _, _, deps = first.partition(": ") + return deps.split() + + +def _deps_in(dep_file): + """The same prerequisites, normalised to paths relative to c/.""" + return {os.path.normpath(d) for d in _raw_deps(dep_file)} + + +def _unconditional_includes(source): + """`#include "..."` lines of a source that no #if/#ifdef encloses.""" + depth, found = 0, [] + for line in source.read_text(encoding="utf-8", errors="replace").splitlines(): + if COND_OPEN_RE.match(line): + depth += 1 + elif COND_CLOSE_RE.match(line): + depth = max(0, depth - 1) + elif depth == 0: + m = INCLUDE_RE.match(line) + if m: + found.append(m.group(1)) + return found + + +def _rule_form(target, exe): + """How a built target is spelled in the Makefile: colibri.exe -> colibri$(EXE).""" + if target.endswith(".o"): + return target + if exe and target.endswith(exe): + return target[:-len(exe)] + "$(EXE)" + return target + "$(EXE)" + + +def _source_of(form, rules_text): + """The .c a rule compiles: the first source among its prerequisites.""" + for m in re.finditer(r"(?m)^" + re.escape(form) + r":[ \t]*(.*)$", rules_text): + for word in m.group(1).split("#", 1)[0].split(): + if SOURCE_RE.search(word): + return C_DIR / word + return None + + +class EngineListTest(unittest.TestCase): def test_the_engine_list_is_derived_and_not_empty(self): - """A parse that matches nothing would make every other check vacuous. - - The rule syntax is the input to this whole file. If it ever changes, - the header check below starts passing for zero targets and says - nothing, which reads exactly like a clean tree. The registry is the - authority on what must be buildable, so cross-check against it rather - than against a second list kept here. - """ + """A parse that matches nothing would make every other check vacuous.""" rules = _engine_rules() self.assertTrue(rules, "no `NAME$(EXE):` rule with a matching NAME.c was " "found -- the Makefile rule syntax changed and " @@ -84,76 +200,156 @@ def test_the_engine_list_is_derived_and_not_empty(self): f"{family.id}: registered family whose build target is not " f"a checkable `{family.build_target}$(EXE):` rule") - def test_every_engine_rule_lists_fp8_format_h_if_it_reaches_it(self): - """A prerequisite reached THROUGH another header is still a prerequisite. - - The check below walks only an engine's own `#include "..."` lines, so a - header pulled in transitively is invisible to it. fp8_format.h is - exactly that shape: quant.h includes it, no engine .c does, so the - direct check stays green whether or not a rule lists it -- while - `touch fp8_format.h; make colibri` would report success and leave a - stale binary carrying the previous FP8_BLOCK. That is the drift the - header's own comment says it exists to prevent, so it is pinned here. - - Bite: delete `fp8_format.h` from `colibri$(EXE):` and this fails naming - colibri; the direct check above still passes. - - Scoped to fp8_format.h ON PURPOSE. Running the same closure over every - header reports seven rules that predate this change and are unrelated - to it -- colibri and deepseek_v41 do not list tok_unicode_o200k.h, - five engines do not list decode_batch.h or edge_adapter_internal.h, - qwen36 does not list sse41_kernels.h. Those are real instances of the - same hazard and worth a separate fix; widening this test to cover them - here would make it fail on arrival for reasons this change did not - cause. - """ + +@unittest.skipUnless(MAKE, "make is not installed") +class GeneratedDepsWiringTest(unittest.TestCase): + """What the Makefile asks for, checked without compiling anything.""" + + @classmethod + def setUpClass(cls): + cls.exe = (_make_var("EXE") or [""])[0] + cls.bins = _make_var("AUTODEP_BINS") + cls.multi = _make_var("AUTODEP_MULTI_TU") + cls.engines = _make_var("ENGINE_RULES") + + def test_every_engine_gets_generated_deps(self): + self.assertTrue(self.bins, "AUTODEP_BINS is empty") + for name in _engine_rules(): + with self.subTest(engine=name): + self.assertIn(name + self.exe, self.bins, + f"{name}: an engine rule outside AUTODEP_BINS keeps " + f"no dependencies at all once its list is gone") + self.assertEqual(sorted(self.engines), sorted(_engine_rules()), + "the Makefile's ENGINE_RULES and this file's derivation " + "disagree about which rules are engines") + + def test_no_generated_rule_lists_headers_by_hand(self): + """A hand-listed header on these rules is redundant and brings back the + merge conflicts #1741 removed. Bite: add `st.h` to `colibri$(EXE):`.""" + text = _joined_makefile() problems = [] - for target, (source, prereqs) in sorted(_engine_rules().items()): - seen, stack = set(), [source] - while stack: - cur = stack.pop() - try: - text = cur.read_text(encoding="utf-8") - except (OSError, UnicodeDecodeError): - continue - for h in INCLUDE_RE.findall(text): - if h in seen: - continue - path = C_DIR / h - # Only headers that exist in c/ are ours to track; system - # headers and generated files are not prerequisites. - if path.exists(): - seen.add(h) - stack.append(path) - if "fp8_format.h" in seen and "fp8_format.h" not in prereqs: - problems.append(f"{target} ({source.name}) reaches fp8_format.h " - f"through its include graph but does not list it") - self.assertEqual( - problems, [], - "fp8_format.h is reachable from these engines but absent from their " - "Makefile prerequisites. Editing it alone will NOT relink them -- " - "make reports success and leaves a stale artifact:\n " - + "\n ".join(problems), - ) - - def test_every_engine_rule_lists_the_headers_its_source_includes(self): + for target in self.bins: + form = _rule_form(target, self.exe) + for m in re.finditer(r"(?m)^" + re.escape(form) + r":[ \t]*(.*)$", text): + listed = [w for w in m.group(1).split("#", 1)[0].split() + if re.search(r"\.(h|inc)$", w) and not w.startswith("$")] + if listed: + problems.append(f"{form}: {' '.join(listed)}") + self.assertEqual(problems, [], + "these rules get their headers from -MMD and should " + "not list them by hand:\n " + "\n ".join(problems)) + + def _one_unit_problems(self, *variables): + commands = _commands(self.bins, *variables) problems = [] - for target, (source, prereqs) in sorted(_engine_rules().items()): - included = set(INCLUDE_RE.findall(source.read_text(encoding="utf-8"))) - # Only headers that exist in c/ are ours to track; system headers - # and anything generated elsewhere are not prerequisites. - local = {h for h in included if (C_DIR / h).exists()} - missing = sorted(local - prereqs) - if missing: - problems.append(f"{target} ({source.name}) is missing: " - f"{' '.join(missing)}") - self.assertEqual( - problems, [], - "Headers included by an engine but absent from its Makefile " - "prerequisites. Editing one of these alone will NOT relink the " - "binary -- make will report success and leave a stale artifact " - ":\n " + "\n ".join(problems), - ) + for target in self.bins: + words = commands.get(target) + if words is None: + problems.append(f"{target}: make printed no command producing it") + continue + units = sorted({w for w in words if SOURCE_RE.search(w)}) + if "-MMD" not in words: + problems.append(f"{target}: compiled without -MMD, so no .d is " + f"written (is it built with $(CFLAGS)?)") + elif len(units) != 1: + problems.append(f"{target}: compiles {len(units)} units in one " + f"command ({' '.join(units)}); GCC writes only the " + f"last one's dependencies. Compile to objects, or " + f"add it to AUTODEP_MULTI_TU and list its headers") + return problems + + def test_every_generated_rule_compiles_one_unit_with_mmd(self): + """Point 1 of the module docstring, default build. Bite: make a test's + recipe compile `$< segment_runtime.c` and leave it out of + AUTODEP_MULTI_TU.""" + problems = self._one_unit_problems() + self.assertEqual(problems, [], "\n " + "\n ".join(problems)) + + def test_every_generated_rule_compiles_one_unit_in_the_gpu_build(self): + """The same with CUDA/HIP on, where the qwen36 tier joins the build. + Bite: put qwen36_tier.c back on qwen36's command line.""" + variable = _gpu_variables(self.exe) + if variable is None: + self.skipTest("this host emits no GPU recipe for colibri") + problems = self._one_unit_problems(variable) + self.assertEqual(problems, [], f"with {variable}:\n " + "\n ".join(problems)) + + def test_multi_unit_exceptions_are_still_multi_unit(self): + """An exception that no longer applies should rejoin AUTODEP_BINS.""" + targets = [f"tests/{n}{self.exe}" for n in self.multi] + seen = _commands(targets) + stale = [t for t in targets + if len({w for w in seen.get(t, []) if SOURCE_RE.search(w)}) < 2] + self.assertEqual(stale, [], "these now compile a single unit; move them " + "out of AUTODEP_MULTI_TU and drop their " + "header lists") + + +@unittest.skipUnless(MAKE, "make is not installed") +class GeneratedDepsCoverageTest(unittest.TestCase): + """After a build: every built target's .d exists and covers its includes.""" + + @classmethod + def setUpClass(cls): + cls.exe = (_make_var("EXE") or [""])[0] + cls.built = [t for t in _make_var("AUTODEP_BINS") if (C_DIR / t).exists()] + if not cls.built: + raise unittest.SkipTest("nothing in AUTODEP_BINS is built here; run " + "this after `make check`") + + def test_every_built_target_has_a_dep_file_covering_its_includes(self): + text = _joined_makefile() + problems = [] + for target in self.built: + dep = _dep_file(target) + if not dep.exists(): + problems.append(f"{target}: built, but has no {dep.name}; its " + f"headers are untracked until it is rebuilt") + continue + source = _source_of(_rule_form(target, self.exe), text) + if source is None: + problems.append(f"{target}: no source found in its rule") + continue + deps = _deps_in(dep) + for inc in _unconditional_includes(source): + path = source.parent / inc + if not path.exists(): + continue + rel = os.path.normpath(os.path.relpath(path, C_DIR)) + if rel not in deps: + problems.append(f"{target}: {dep.name} does not name {rel}, " + f"which {source.name} includes") + self.assertEqual(problems, [], "\n " + "\n ".join(problems)) + + def test_make_rebuilds_a_target_when_its_header_changes(self): + """The end-to-end form of #1284's reproduction, with nothing touched: + `make -q -W
` pretends the header is newer. Engines first, then + tests, since `make check` builds every test but few engines. Bite: + remove the `-include` line at the end of the Makefile and every target + answers "up to date".""" + engines = [n + self.exe for n in _engine_rules()] + candidates = [t for t in engines if t in self.built] + \ + [t for t in self.built if t.startswith("tests/")][:6] + checked, problems = 0, [] + for target in candidates: + dep = _dep_file(target) + if not dep.exists(): + continue + # `make check` builds colibri with ARCH=portable and runs the + # tests with the default ARCH, so .build-config can legitimately + # make a target out of date already. A -W answer means nothing then. + if _make("-q", target).returncode != 0: + continue + # -W matches a prerequisite by name, so use the .d's own spelling. + headers = [h for h in _raw_deps(dep) if h.endswith(".h")][:2] + for header in headers: + checked += 1 + if _make("-q", "-W", header, target).returncode != 1: + problems.append(f"{target}: `make -q -W {header}` says up to " + f"date, though {dep.name} names it") + if not checked: + self.skipTest("no target is built and up to date here") + self.assertEqual(problems, [], "\n " + "\n ".join(problems)) if __name__ == "__main__": diff --git a/c/tools/clean.py b/c/tools/clean.py index 06a588b7f..ad0a71b95 100644 --- a/c/tools/clean.py +++ b/c/tools/clean.py @@ -25,7 +25,7 @@ "glm53", "glm53.exe", "glm", "glm.exe", # pre-rename name of the colibri engine "iobench", "iobench.exe", - "backend_cuda.o", "backend_loader.o", + "backend_cuda.o", "backend_loader.o", "qwen36_tier.o", "backend_cuda_test", "backend_cuda_test.exe", "mxfp4_expert_cuda_test", "mxfp4_expert_cuda_test.exe", "backend_cuda_bench", "backend_cuda_bench.exe", @@ -59,8 +59,12 @@ # A stale probe does not just waste space -- it is the owner that PRODUCES # physical execution evidence, and a stale one reports PASS for code that is no # longer in the tree. +# +# *.d are the dependency files -MMD writes beside each binary and object +# (#1741). A stale one only adds prerequisites, but clean should leave nothing +# the build made, and removing it forces the rebuild that writes a fresh one. ARTIFACT_GLOBS = ["tests/test_*", "tests/bench_*", "tests/fuzz_*", - "tests/*_probe*", "COLI_V4_UNIT_*.o"] + "tests/*_probe*", "COLI_V4_UNIT_*.o", "*.d", "tests/*.d"] KEEP_EXT = (".c", ".h", ".cc", ".cpp", ".cu", ".mm", ".py", ".txt", ".json", ".md", ".bin", ".sh", ".toml", ".yml", ".yaml") # Directories to remove. From ebdb59130e135ba049df70d260694717a96666ec Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 14:28:52 +0200 Subject: [PATCH 2/7] test: follow the qwen36 tier variable to its new name test_registry_engine_agreement treats an engine as accelerated when its rule names a backend object. QWEN36_TIER_SRC became QWEN36_TIER_OBJ in the previous commit, so the old entry matched nothing; qwen36 and qwen38 still passed through CUDA_OBJ, which hid it. --- c/tests/test_registry_engine_agreement.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/c/tests/test_registry_engine_agreement.py b/c/tests/test_registry_engine_agreement.py index d90e931c7..7074112d1 100644 --- a/c/tests/test_registry_engine_agreement.py +++ b/c/tests/test_registry_engine_agreement.py @@ -34,7 +34,7 @@ BACKEND_OBJECTS = ("CUDA_OBJ", "METAL_OBJ", "VK_OBJ", "VK_SPV", "INK_CUDA_OBJ", - "QWEN36_TIER_SRC") + "QWEN36_TIER_OBJ") def engine_rule(artifact: str) -> str: From 081d28dd4f9fb0c1560c5825b1aec6d39457ae80 Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 17:16:40 +0200 Subject: [PATCH 3/7] build: address review of the generated-deps change - test_makefile_deps: every make call keeps .build-config's contents and timestamp. Parsing with CUDA=1 or CUDA_DLL=1 rewrote it, so running the suite switched the tree's recorded configuration and the next make relinked everything that depends on it. A new test pins this. - clean.py: remove the .d files under tools/ and build/segment/ too, and the backend_vulkan.o and backend_xdna.o objects. After a clean that removed their .d, those objects stayed in the tree with no record of the headers they read, and the coverage test rightly flagged them. - One sed pass over the rule names now serves both TEST_RULES and ENGINE_RULES; both variables are unchanged. - A stale comment still said the qwen36 tests compile qwen36_tier.c; the deepseek_v41 rule loses a leftover line continuation. --- c/Makefile | 14 +++++---- c/tests/test_makefile_deps.py | 59 ++++++++++++++++++++++++++++------- c/tools/clean.py | 9 +++++- 3 files changed, 64 insertions(+), 18 deletions(-) diff --git a/c/Makefile b/c/Makefile index 4aaf6e6c6..95d966cd0 100644 --- a/c/Makefile +++ b/c/Makefile @@ -496,7 +496,10 @@ endif # in the other direction: it promoted files that deliberately have no rule (a branch's # work-in-progress test, test_fse, test_tok, test_tok_kimi, test_vk_mxfp4) into gates, and # they fail to link. Having a rule is the honest definition of "this is a gate". -TEST_RULES := $(shell sed -n 's|^tests/\(test_[a-z0-9_]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST))) +# RULE_NAMES is every `NAME$(EXE):` rule, with its directory; one pass over the +# Makefile serves both TEST_RULES here and ENGINE_RULES (#1741, end of file). +RULE_NAMES := $(shell sed -n 's|^\([a-z0-9_/]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST))) +TEST_RULES := $(patsubst tests/%,%,$(filter tests/test_%,$(RULE_NAMES))) # test_uring is Linux-only. V4 engine tests are appended below only on supported # x86-64 Linux/Windows and aarch64 Linux hosts; the V4 infrastructure tests have # unconditional rules and therefore remain portable gates. The six forced @@ -1206,8 +1209,7 @@ qwen36$(EXE): qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) .build-config # DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts # stream from the official checkpoint and quant.h's mxfp4 kernel reads them as they # are, so there is no conversion target here to go with it. -deepseek_v41$(EXE): deepseek_v41.c \ - .build-config +deepseek_v41$(EXE): deepseek_v41.c .build-config $(CC) $(CFLAGS) deepseek_v41.c -o deepseek_v41$(EXE) $(LDFLAGS) # Qwen3.8-Flash-Next text-only sibling. qwen38.c owns the direct checkpoint @@ -1266,8 +1268,8 @@ qwen38-tier-engine-check: qwen38-tiny-fp8-generate tests/test_qwen38_tier_engine # Context-size gates: KV layout, growth across requests, and the attention # capacity. Includes qwen36.c directly, so no model file is needed. -# Same tier sources as the engine: the test includes qwen36.c, so with CUDA=1 -# it needs qwen36_tier.c and the backend object too (without CUDA, the header's +# Same tier objects as the engine: the test includes qwen36.c, so with CUDA=1 +# it links qwen36_tier.o and the backend object too (without CUDA, the header's # inline stubs cover it and both are empty). tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c qwen36.c $(QWEN36_TIER_OBJ) $(CUDA_OBJ) $(CC) $(CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS) @@ -2465,7 +2467,7 @@ AUTODEP_MULTI_TU = test_xdna_qt_state test_xdna_failure test_qwen38_tokenizer \ AUTODEP_ELSEWHERE = test_segment_adapters_registration test_segment_adapters_real \ test_edge_adapters_registration test_edge_adapters_real test_deepseek_v4 \ test_v4_serve_framing -ENGINE_RULES := $(foreach n,$(shell sed -n 's|^\([a-z0-9_]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST))),$(if $(wildcard $(n).c),$(n))) +ENGINE_RULES := $(foreach n,$(RULE_NAMES),$(if $(findstring /,$(n)),,$(if $(wildcard $(n).c),$(n)))) AUTODEP_OBJS = backend_vulkan.o backend_loader.o backend_xdna.o qwen36_tier.o AUTODEP_BINS = $(addsuffix $(EXE),$(ENGINE_RULES)) \ $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_MULTI_TU) $(AUTODEP_ELSEWHERE),$(TEST_RULES)))) \ diff --git a/c/tests/test_makefile_deps.py b/c/tests/test_makefile_deps.py index 8462533da..4559e388a 100644 --- a/c/tests/test_makefile_deps.py +++ b/c/tests/test_makefile_deps.py @@ -34,6 +34,7 @@ matching `NAME$(EXE):` with a `NAME.c` beside it is an engine, and the family registry is the cross-check that the parse found them all. """ +import contextlib import os import re import shlex @@ -72,24 +73,47 @@ def _engine_rules(): return rules -def _make(*args): - """Run make in c/ with the build's own configuration. +@contextlib.contextmanager +def _build_config_kept(): + """Leave .build-config exactly as it was, contents and timestamp. - Inside `make check` the child inherits MAKEFLAGS, so it parses the same - variables the build did and leaves .build-config as it was. + The Makefile rewrites .build-config while it parses, whenever the flags + differ from the last build's. Every make call below parses, and some pass + CUDA=1 or CUDA_DLL=1 on purpose, so without this the suite would switch + the tree's recorded configuration and the next `make` would relink + everything that depends on it. """ - return subprocess.run([MAKE, "--no-print-directory", *args], cwd=C_DIR, - text=True, encoding="utf-8", errors="replace", - capture_output=True, timeout=600) + path = C_DIR / ".build-config" + before = path.read_bytes() if path.exists() else None + stamp = path.stat() if before is not None else None + try: + yield + finally: + if before is None: + if path.exists(): + path.unlink() + elif not path.exists() or path.read_bytes() != before or \ + path.stat().st_mtime_ns != stamp.st_mtime_ns: + path.write_bytes(before) + os.utime(path, ns=(stamp.st_atime_ns, stamp.st_mtime_ns)) + + +def _make(*args): + """Run make in c/, leaving the tree's build configuration untouched.""" + with _build_config_kept(): + return subprocess.run([MAKE, "--no-print-directory", *args], cwd=C_DIR, + text=True, encoding="utf-8", errors="replace", + capture_output=True, timeout=600) def _make_var(name): # `make --eval` needs GNU Make 3.82; macOS ships 3.81, which does read an # extra makefile from stdin. - proc = subprocess.run([MAKE, "-s", "--no-print-directory", "-f", "Makefile", - "-f", "-", f"print-{name}"], cwd=C_DIR, text=True, - input="print-%: ; @echo $($*)\n", capture_output=True, - timeout=300) + with _build_config_kept(): + proc = subprocess.run([MAKE, "-s", "--no-print-directory", "-f", "Makefile", + "-f", "-", f"print-{name}"], cwd=C_DIR, text=True, + input="print-%: ; @echo $($*)\n", capture_output=True, + timeout=300) if proc.returncode != 0: raise AssertionError(f"could not read {name} from the Makefile:\n" f"{proc.stderr}") @@ -274,6 +298,19 @@ def test_every_generated_rule_compiles_one_unit_in_the_gpu_build(self): problems = self._one_unit_problems(variable) self.assertEqual(problems, [], f"with {variable}:\n " + "\n ".join(problems)) + def test_the_checks_leave_the_build_config_alone(self): + """Parsing with CUDA=1 or CUDA_DLL=1 rewrites .build-config; the checks + above do exactly that, and must not change which configuration the + tree says it was built with. Bite: drop the `with _build_config_kept()` + from `_make` and this fails whenever a GPU flavour is accepted here.""" + path = C_DIR / ".build-config" + before = (path.read_bytes(), path.stat().st_mtime_ns) if path.exists() else None + _gpu_variables(self.exe) + _make("-Bn", "colibri" + self.exe, "XDNA=1") + after = (path.read_bytes(), path.stat().st_mtime_ns) if path.exists() else None + self.assertEqual(after, before, ".build-config was changed by a make call " + "that only meant to read the Makefile") + def test_multi_unit_exceptions_are_still_multi_unit(self): """An exception that no longer applies should rejoin AUTODEP_BINS.""" targets = [f"tests/{n}{self.exe}" for n in self.multi] diff --git a/c/tools/clean.py b/c/tools/clean.py index ad0a71b95..dc1fcd8b0 100644 --- a/c/tools/clean.py +++ b/c/tools/clean.py @@ -26,6 +26,9 @@ "glm", "glm.exe", # pre-rename name of the colibri engine "iobench", "iobench.exe", "backend_cuda.o", "backend_loader.o", "qwen36_tier.o", + # VK=1 and XDNA=1 objects. Left behind once their .d is cleaned, an + # object would sit in the tree with no record of the headers it read. + "backend_vulkan.o", "backend_xdna.o", "backend_cuda_test", "backend_cuda_test.exe", "mxfp4_expert_cuda_test", "mxfp4_expert_cuda_test.exe", "backend_cuda_bench", "backend_cuda_bench.exe", @@ -63,8 +66,12 @@ # *.d are the dependency files -MMD writes beside each binary and object # (#1741). A stale one only adds prerequisites, but clean should leave nothing # the build made, and removing it forces the rebuild that writes a fresh one. +# They land wherever an output does: c/ and tests/ for the engines and tests, +# tools/ for the ctypes libraries, build/segment/ for the V4 unit objects +# (build/ownership/ goes as a whole directory below). ARTIFACT_GLOBS = ["tests/test_*", "tests/bench_*", "tests/fuzz_*", - "tests/*_probe*", "COLI_V4_UNIT_*.o", "*.d", "tests/*.d"] + "tests/*_probe*", "COLI_V4_UNIT_*.o", "*.d", "tests/*.d", + "tools/*.d", "build/segment/*.d"] KEEP_EXT = (".c", ".h", ".cc", ".cpp", ".cu", ".mm", ".py", ".txt", ".json", ".md", ".bin", ".sh", ".toml", ".yml", ".yaml") # Directories to remove. From d9ad16b534d9d2639d43dd323afa32c7472cfb8d Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 17:21:57 +0200 Subject: [PATCH 4/7] build: compile the tests' helper sources to objects Eight tests compiled a second .c on their own command line (segment_runtime.c, edge_runtime.c, backend_xdna.c, deepseek_v4.c and a fixtures file). GCC writes a .d for the last unit only, so they kept hand-written header lists as exceptions, and those lists had the same kind of gaps as the ones -MMD replaced elsewhere. Each helper is now its own object in tests/, compiled with exactly the flags that test gave it before: -DCOLI_XDNA for the XDNA lane object, -DCOLI_V4_UNIT_NATIVE_QUANT for deepseek_v4.c, and NOCUDA_CFLAGS for the two qwen38 tests, which get their own copies of segment_runtime.o and edge_runtime.o since those flags differ from CFLAGS in GPU builds. Each unit was already compiled on its own, so the result is the same. AUTODEP_MULTI_TU is gone, and with it the last hand-written header lists on gcc-built tests; the one-unit check in test_makefile_deps.py now points at objects instead of an exception list. clean.py removes tests/*.o. --- c/Makefile | 88 +++++++++++++++++++++++------------ c/tests/test_makefile_deps.py | 28 ++++------- c/tools/clean.py | 4 +- 3 files changed, 70 insertions(+), 50 deletions(-) diff --git a/c/Makefile b/c/Makefile index 95d966cd0..816f11451 100644 --- a/c/Makefile +++ b/c/Makefile @@ -1371,8 +1371,12 @@ tests/test_ue8m0$(EXE): tests/test_ue8m0.c # Compiles the engine with -DCOLI_XDNA so the QT side pointer, its reset helper # and the expert-slot wiring are exercised as the engine actually builds them. -tests/test_xdna_qt_state$(EXE): tests/test_xdna_qt_state.c colibri.c backend_xdna.c backend_xdna.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) -DCOLI_XDNA $< backend_xdna.c -o $@ $(LDFLAGS) +tests/test_xdna_qt_state$(EXE): tests/test_xdna_qt_state.c colibri.c tests/backend_xdna_lane.o + $(CC) $(CFLAGS) -DCOLI_XDNA $< tests/backend_xdna_lane.o -o $@ $(LDFLAGS) +# backend_xdna.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/backend_xdna_lane.o: backend_xdna.c .build-config + $(CC) $(CFLAGS) -DCOLI_XDNA -c backend_xdna.c -o $@ tests/test_xdna_prepared_state$(EXE): tests/test_xdna_prepared_state.c backend_xdna.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1398,8 +1402,8 @@ tests/test_xdna_execution$(EXE): tests/test_xdna_execution.c backend_xdna.c $(XD # Failure/fallback owner. Needs BOTH the engine (matmul_qt is the fallback # target it verifies against) and the synthetic ABI-2 helpers, which is why it # is a separate owner from test_xdna_execution. -tests/test_xdna_failure$(EXE): tests/test_xdna_failure.c colibri.c backend_xdna.c backend_xdna.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h $(XDNA_FAKE_HELPERS) - $(CC) $(CFLAGS) -DCOLI_XDNA $< backend_xdna.c -o $@ $(LDFLAGS) +tests/test_xdna_failure$(EXE): tests/test_xdna_failure.c colibri.c tests/backend_xdna_lane.o $(XDNA_FAKE_HELPERS) + $(CC) $(CFLAGS) -DCOLI_XDNA $< tests/backend_xdna_lane.o -o $@ $(LDFLAGS) # NOT a test gate -- intentionally absent from TEST_BINS (its name does not match # the tests/test_*$(EXE) pattern the gate list is derived from). It needs a real @@ -1436,8 +1440,12 @@ tests/test_st_pread$(EXE): tests/test_st_pread.c tests/test_st_slice$(EXE): tests/test_st_slice.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h - $(CC) $(NOCUDA_CFLAGS) $< edge_runtime.c -o $@ $(NOCUDA_LDFLAGS) +tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c qwen38.c tests/edge_runtime_nocuda.o + $(CC) $(NOCUDA_CFLAGS) $< tests/edge_runtime_nocuda.o -o $@ $(NOCUDA_LDFLAGS) +# edge_runtime.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/edge_runtime_nocuda.o: edge_runtime.c .build-config + $(CC) $(NOCUDA_CFLAGS) -c edge_runtime.c -o $@ tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) @@ -1476,8 +1484,12 @@ tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c qwen38.c tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c qwen38.c $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h - $(CC) $(NOCUDA_CFLAGS) $< segment_runtime.c -o $@ $(NOCUDA_LDFLAGS) +tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c qwen38.c tests/segment_runtime_nocuda.o + $(CC) $(NOCUDA_CFLAGS) $< tests/segment_runtime_nocuda.o -o $@ $(NOCUDA_LDFLAGS) +# segment_runtime.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/segment_runtime_nocuda.o: segment_runtime.c .build-config + $(CC) $(NOCUDA_CFLAGS) -c segment_runtime.c -o $@ tests/test_st_map$(EXE): tests/test_st_map.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1594,11 +1606,19 @@ tests/test_serve_codec$(EXE): tests/test_serve_codec.c tests/test_serve_budget$(EXE): tests/test_serve_budget.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c segment_runtime.c segment_runtime.h - $(CC) $(CFLAGS) tests/test_segment_runtime.c segment_runtime.c -o $@ $(LDFLAGS) +tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c tests/segment_runtime.o + $(CC) $(CFLAGS) tests/test_segment_runtime.c tests/segment_runtime.o -o $@ $(LDFLAGS) +# segment_runtime.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/segment_runtime.o: segment_runtime.c .build-config + $(CC) $(CFLAGS) -c segment_runtime.c -o $@ -tests/test_segment_conformance$(EXE): tests/test_segment_conformance.c tests/segment_conformance_fixtures.c tests/segment_conformance_fixtures.h segment_runtime.c segment_runtime.h - $(CC) $(CFLAGS) tests/test_segment_conformance.c tests/segment_conformance_fixtures.c segment_runtime.c -o $@ $(LDFLAGS) +tests/test_segment_conformance$(EXE): tests/test_segment_conformance.c tests/segment_conformance_fixtures.o tests/segment_runtime.o + $(CC) $(CFLAGS) tests/test_segment_conformance.c tests/segment_conformance_fixtures.o tests/segment_runtime.o -o $@ $(LDFLAGS) +# tests/segment_conformance_fixtures.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/segment_conformance_fixtures.o: tests/segment_conformance_fixtures.c .build-config + $(CC) $(CFLAGS) -c tests/segment_conformance_fixtures.c -o $@ tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) @@ -1761,11 +1781,19 @@ tests/test_compat_direct$(EXE): tests/test_compat_direct.c tests/test_expert_store_ops$(EXE): tests/test_expert_store_ops.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_native_quant$(EXE): tests/test_native_quant.c sse41_kernels.h deepseek_v4.c native_quant.h tensor.h quant.h fp8_format.h idot.h - $(CC) $(CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT deepseek_v4.c $< -o $@ $(LDFLAGS) +tests/test_native_quant$(EXE): tests/test_native_quant.c tests/deepseek_v4_native_quant.o + $(CC) $(CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT tests/deepseek_v4_native_quant.o $< -o $@ $(LDFLAGS) +# deepseek_v4.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/deepseek_v4_native_quant.o: deepseek_v4.c .build-config + $(CC) $(CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT -c deepseek_v4.c -o $@ -tests/test_edge_runtime$(EXE): tests/test_edge_runtime.c edge_runtime.c edge_runtime.h - $(CC) $(CFLAGS) tests/test_edge_runtime.c edge_runtime.c -o $@ $(LDFLAGS) +tests/test_edge_runtime$(EXE): tests/test_edge_runtime.c tests/edge_runtime.o + $(CC) $(CFLAGS) tests/test_edge_runtime.c tests/edge_runtime.o -o $@ $(LDFLAGS) +# edge_runtime.c as its own object, compiled with the flags this test gave it on +# the same command line before (#1741): one unit per command keeps each .d whole. +tests/edge_runtime.o: edge_runtime.c .build-config + $(CC) $(CFLAGS) -c edge_runtime.c -o $@ ifeq ($(COLI_V4_SUPPORTED),1) include Makefile.deepseek-v4.units @@ -2447,30 +2475,32 @@ tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_ # does not list headers by hand. The engines and tests are derived the same way # TEST_RULES is, so a new rule joins automatically. # -# Two kinds of rule keep hand-written header lists instead: -# - AUTODEP_MULTI_TU compiles more than one .c in a single command. GCC then -# writes only the LAST unit's dependencies, so a .d would silently miss the -# headers of the others. tests/test_makefile_deps.py fails if a recipe of -# that shape appears anywhere in AUTODEP_BINS. -# - AUTODEP_ELSEWHERE is built outside this Makefile's own $(CC) recipes -# (Makefile.deepseek-v4 and the adapter test targets), as are the nvcc, -# MSVC-hosted CUDA and Metal objects, which are not listed here at all. +# A command that compiles several .c files writes a .d for the LAST one only, +# silently dropping the headers of the others. So a helper source a test links +# is its own object (the tests/*.o rules, each next to its test), compiled with +# the flags that test gives it. tests/test_makefile_deps.py fails if a rule in +# AUTODEP_BINS ever compiles more than one unit. +# +# AUTODEP_ELSEWHERE is built outside this Makefile's own $(CC) recipes +# (Makefile.deepseek-v4 and the adapter test targets) and keeps its header +# lists, as do the nvcc, MSVC-hosted CUDA and Metal objects, which are not +# listed here at all. # # Most rules do not depend on .build-config, so a change to CFLAGS alone would # not rebuild a binary made before this change, and it would have no .d. Each # target therefore also depends on its own .d through an empty rule: a missing # .d counts as out of date, the target is rebuilt, and that compile writes it. # The same covers a .d deleted by hand. -AUTODEP_MULTI_TU = test_xdna_qt_state test_xdna_failure test_qwen38_tokenizer \ - test_qwen38_native_weights test_segment_runtime test_segment_conformance \ - test_native_quant test_edge_runtime AUTODEP_ELSEWHERE = test_segment_adapters_registration test_segment_adapters_real \ test_edge_adapters_registration test_edge_adapters_real test_deepseek_v4 \ test_v4_serve_framing ENGINE_RULES := $(foreach n,$(RULE_NAMES),$(if $(findstring /,$(n)),,$(if $(wildcard $(n).c),$(n)))) -AUTODEP_OBJS = backend_vulkan.o backend_loader.o backend_xdna.o qwen36_tier.o +AUTODEP_OBJS = backend_vulkan.o backend_loader.o backend_xdna.o qwen36_tier.o \ + tests/backend_xdna_lane.o tests/edge_runtime.o tests/edge_runtime_nocuda.o \ + tests/segment_runtime.o tests/segment_runtime_nocuda.o \ + tests/segment_conformance_fixtures.o tests/deepseek_v4_native_quant.o AUTODEP_BINS = $(addsuffix $(EXE),$(ENGINE_RULES)) \ - $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_MULTI_TU) $(AUTODEP_ELSEWHERE),$(TEST_RULES)))) \ + $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_ELSEWHERE),$(TEST_RULES)))) \ $(AUTODEP_OBJS) AUTODEP_DEPS = $(addsuffix .d,$(basename $(AUTODEP_BINS))) $(foreach b,$(AUTODEP_BINS),$(eval $(b): $(basename $(b)).d)) diff --git a/c/tests/test_makefile_deps.py b/c/tests/test_makefile_deps.py index 4559e388a..28adacd85 100644 --- a/c/tests/test_makefile_deps.py +++ b/c/tests/test_makefile_deps.py @@ -15,9 +15,10 @@ 1. A recipe that compiles more than one .c in one command. GCC then writes only the LAST unit's dependencies, so the others' headers are missing from - the .d with nothing to show for it. Such recipes keep hand-written lists - (AUTODEP_MULTI_TU) and every rule in AUTODEP_BINS is checked to compile - exactly one unit, with -MMD, by asking make for the commands it would run. + the .d with nothing to show for it. A helper source a test links is + therefore its own object (the tests/*.o rules), and every rule in + AUTODEP_BINS is checked to compile exactly one unit, with -MMD, by asking + make for the commands it would run. 2. A binary built without a .d, e.g. before this change. Most rules do not depend on .build-config, so nothing else would force the rebuild that writes @@ -233,7 +234,6 @@ class GeneratedDepsWiringTest(unittest.TestCase): def setUpClass(cls): cls.exe = (_make_var("EXE") or [""])[0] cls.bins = _make_var("AUTODEP_BINS") - cls.multi = _make_var("AUTODEP_MULTI_TU") cls.engines = _make_var("ENGINE_RULES") def test_every_engine_gets_generated_deps(self): @@ -278,14 +278,13 @@ def _one_unit_problems(self, *variables): elif len(units) != 1: problems.append(f"{target}: compiles {len(units)} units in one " f"command ({' '.join(units)}); GCC writes only the " - f"last one's dependencies. Compile to objects, or " - f"add it to AUTODEP_MULTI_TU and list its headers") + f"last one's dependencies. Compile the extra " + f"sources to objects, as the tests/*.o rules do") return problems def test_every_generated_rule_compiles_one_unit_with_mmd(self): - """Point 1 of the module docstring, default build. Bite: make a test's - recipe compile `$< segment_runtime.c` and leave it out of - AUTODEP_MULTI_TU.""" + """Point 1 of the module docstring, default build. Bite: put + segment_runtime.c back on test_segment_runtime's command line.""" problems = self._one_unit_problems() self.assertEqual(problems, [], "\n " + "\n ".join(problems)) @@ -311,17 +310,6 @@ def test_the_checks_leave_the_build_config_alone(self): self.assertEqual(after, before, ".build-config was changed by a make call " "that only meant to read the Makefile") - def test_multi_unit_exceptions_are_still_multi_unit(self): - """An exception that no longer applies should rejoin AUTODEP_BINS.""" - targets = [f"tests/{n}{self.exe}" for n in self.multi] - seen = _commands(targets) - stale = [t for t in targets - if len({w for w in seen.get(t, []) if SOURCE_RE.search(w)}) < 2] - self.assertEqual(stale, [], "these now compile a single unit; move them " - "out of AUTODEP_MULTI_TU and drop their " - "header lists") - - @unittest.skipUnless(MAKE, "make is not installed") class GeneratedDepsCoverageTest(unittest.TestCase): """After a build: every built target's .d exists and covers its includes.""" diff --git a/c/tools/clean.py b/c/tools/clean.py index dc1fcd8b0..6805529cd 100644 --- a/c/tools/clean.py +++ b/c/tools/clean.py @@ -71,7 +71,9 @@ # (build/ownership/ goes as a whole directory below). ARTIFACT_GLOBS = ["tests/test_*", "tests/bench_*", "tests/fuzz_*", "tests/*_probe*", "COLI_V4_UNIT_*.o", "*.d", "tests/*.d", - "tools/*.d", "build/segment/*.d"] + "tools/*.d", "build/segment/*.d", + # helper objects the tests link (#1741), one unit per command + "tests/*.o"] KEEP_EXT = (".c", ".h", ".cc", ".cpp", ".cu", ".mm", ".py", ".txt", ".json", ".md", ".bin", ".sh", ".toml", ".yml", ".yaml") # Directories to remove. From 45dfb4df877330d68076c8d99a0edd13e1367774 Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 17:24:12 +0200 Subject: [PATCH 5/7] test: check the one-unit rule in every build flavour Headers under #if reach a .d only through the compile that reads them: backend_xdna.h with XDNA=1, backend_vulkan.h with VK=1, the qwen36 tier only with CUDA or HIP. The coverage test cannot require them from the source text, so what stands behind them is the one-unit check, and that only ran in the default and one GPU configuration. It now runs in every flavour the host's make accepts: CUDA=1, CUDA_DLL=1, HIP=1 (with a fixed HIP_ARCH, as the CI syntax job uses), HIP_DLL=1, VK=1, XDNA=1 and METAL=1, each recognised by the define colibri then gets. On Windows that is CUDA_DLL, HIP_DLL, VK and XDNA; on Linux CUDA, HIP, VK and XDNA. Putting backend_xdna.c on colibri's command line under XDNA=1 passes the default check and fails this one, in XDNA=1 only. --- c/tests/test_makefile_deps.py | 74 +++++++++++++++++++++++------------ 1 file changed, 48 insertions(+), 26 deletions(-) diff --git a/c/tests/test_makefile_deps.py b/c/tests/test_makefile_deps.py index 28adacd85..741f3a457 100644 --- a/c/tests/test_makefile_deps.py +++ b/c/tests/test_makefile_deps.py @@ -28,8 +28,11 @@ 3. A .d that does not cover what its source includes. Every unconditional `#include "..."` of each built target's source must appear in its .d, and for the engines `make -q -W
` must actually answer "rebuild". - Includes under #if are not required: `uring.h` is read on Linux only, and a - Windows build that does not depend on it is correct. + Includes under #if are not required here: `uring.h` is read on Linux only, + and a Windows build that does not depend on it is correct. What stands + behind them instead is point 1, checked in every build flavour the host + accepts: a compile that reads a conditional header is a one-unit compile + with -MMD, so the header is in that flavour's .d. The engine list is still DERIVED from the Makefile, never kept here: anything matching `NAME$(EXE):` with a `NAME.c` beside it is an engine, and the family @@ -141,20 +144,34 @@ def _commands(targets, *variables): return commands -def _gpu_variables(exe): - """The make variable that builds the GPU flavour here, or None. +# Every build switch that changes what a compile reads, with the define that +# shows it took effect. HIP=1 gets a fixed architecture, as the CI syntax job +# does, because the default asks for a GPU the host may not have. +FLAVOURS = ((("CUDA=1",), "-DCOLI_CUDA"), + (("CUDA_DLL=1",), "-DCOLI_CUDA"), + (("HIP=1", "HIP_ARCH=gfx1100"), "-DCOLI_CUDA"), + (("HIP_DLL=1",), "-DCOLI_HIP_DLL"), + (("VK=1",), "-DCOLI_VULKAN"), + (("XDNA=1",), "-DCOLI_XDNA"), + (("METAL=1",), "-DCOLI_METAL")) - QWEN36_TIER_OBJ only exists with CUDA/HIP, and a GPU build is where a - second unit last crept onto a command line, so the one-unit check has to - see that configuration too. CUDA=1 is refused off Linux and the wording - differs per platform, so test the fact rather than the message, the way - test_makefile_cuda_scope does: does colibri then get -DCOLI_CUDA? + +def _accepted_flavours(exe): + """The build flavours this host's make accepts, as variable tuples. + + Headers under #if are read only in the flavour that turns them on + (backend_xdna.h with XDNA=1, backend_vulkan.h with VK=1, qwen36_tier.o + only with CUDA or HIP), so the one-unit check has to see each of them. + Several are refused off their platform, with wording that differs per + platform, so test the fact rather than the message, the way + test_makefile_cuda_scope does: does colibri then get the define? """ - for variable in ("CUDA=1", "CUDA_DLL=1"): - proc = _make("-Bn", "colibri" + exe, variable) - if proc.returncode == 0 and "-DCOLI_CUDA" in proc.stdout: - return variable - return None + accepted = [] + for variables, define in FLAVOURS: + proc = _make("-Bn", "colibri" + exe, *variables) + if proc.returncode == 0 and define in proc.stdout: + accepted.append(variables) + return accepted def _dep_file(target): @@ -288,28 +305,33 @@ def test_every_generated_rule_compiles_one_unit_with_mmd(self): problems = self._one_unit_problems() self.assertEqual(problems, [], "\n " + "\n ".join(problems)) - def test_every_generated_rule_compiles_one_unit_in_the_gpu_build(self): - """The same with CUDA/HIP on, where the qwen36 tier joins the build. - Bite: put qwen36_tier.c back on qwen36's command line.""" - variable = _gpu_variables(self.exe) - if variable is None: - self.skipTest("this host emits no GPU recipe for colibri") - problems = self._one_unit_problems(variable) - self.assertEqual(problems, [], f"with {variable}:\n " + "\n ".join(problems)) + def test_every_generated_rule_compiles_one_unit_in_every_build_flavour(self): + """The same in every flavour this host accepts (VK=1, XDNA=1, the GPU + builds, ...). A header under #if reaches a .d only through the compile + that reads it, so this is what stands behind the conditional includes + the coverage test below cannot require. Bite: put qwen36_tier.c back + on qwen36's command line; the GPU flavour then fails.""" + flavours = _accepted_flavours(self.exe) + if not flavours: + self.skipTest("this host accepts no build flavour beyond the default") + for variables in flavours: + with self.subTest(flavour=" ".join(variables)): + problems = self._one_unit_problems(*variables) + self.assertEqual(problems, [], "\n " + "\n ".join(problems)) def test_the_checks_leave_the_build_config_alone(self): - """Parsing with CUDA=1 or CUDA_DLL=1 rewrites .build-config; the checks + """Parsing with another flavour rewrites .build-config; the checks above do exactly that, and must not change which configuration the tree says it was built with. Bite: drop the `with _build_config_kept()` - from `_make` and this fails whenever a GPU flavour is accepted here.""" + from `_make` and this fails whenever a flavour is accepted here.""" path = C_DIR / ".build-config" before = (path.read_bytes(), path.stat().st_mtime_ns) if path.exists() else None - _gpu_variables(self.exe) - _make("-Bn", "colibri" + self.exe, "XDNA=1") + _accepted_flavours(self.exe) after = (path.read_bytes(), path.stat().st_mtime_ns) if path.exists() else None self.assertEqual(after, before, ".build-config was changed by a make call " "that only meant to read the Makefile") + @unittest.skipUnless(MAKE, "make is not installed") class GeneratedDepsCoverageTest(unittest.TestCase): """After a build: every built target's .d exists and covers its includes.""" From 104d63ce6bc1ba40e6e3b927ed09d9cdbc52440a Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Fri, 25 Sep 2026 18:29:12 +0200 Subject: [PATCH 6/7] build: say why the AUTODEP_ELSEWHERE tests keep their lists The comment said all six are built outside this Makefile's own $(CC) recipes. Only the two DeepSeek V4 tests are; the four segment/edge adapter tests are compiled here, three units in one command. --- c/Makefile | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/c/Makefile b/c/Makefile index 816f11451..3c2404939 100644 --- a/c/Makefile +++ b/c/Makefile @@ -2481,10 +2481,11 @@ tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_ # the flags that test gives it. tests/test_makefile_deps.py fails if a rule in # AUTODEP_BINS ever compiles more than one unit. # -# AUTODEP_ELSEWHERE is built outside this Makefile's own $(CC) recipes -# (Makefile.deepseek-v4 and the adapter test targets) and keeps its header -# lists, as do the nvcc, MSVC-hosted CUDA and Metal objects, which are not -# listed here at all. +# AUTODEP_ELSEWHERE keeps its header lists: test_deepseek_v4 and +# test_v4_serve_framing are built by Makefile.deepseek-v4, and the four +# segment/edge adapter tests still compile three units in one command. The +# nvcc, MSVC-hosted CUDA and Metal objects keep theirs too and are not listed +# here at all. # # Most rules do not depend on .build-config, so a change to CFLAGS alone would # not rebuild a binary made before this change, and it would have no .d. Each From fbeef6aa6734da60fa6b60d841dc4b88b01cf691 Mon Sep 17 00:00:00 2001 From: Kenneth-Javier Ortigoza Date: Sat, 26 Sep 2026 00:32:12 +0200 Subject: [PATCH 7/7] build: generate header prerequisites for the rest of the gcc rules The first slice covered the engines, the tests/test_* rules and the objects they link. The other rules gcc compiles with $(CFLAGS) still listed their headers by hand, and drifted the same way: the benchmarks under tests/, the XDNA probe, the objects of the segment library and of the V4 ownership test, and the rANS ctypes library. - AUTODEP_BINS now takes every rule under tests/, not only tests/test_*, plus $(SEGMENT_ALL_OBJS), $(SEGMENT_RUNTIME_OBJS), $(V4_OWNERSHIP_OBJS) and $(RANSLIB). Their header lists are gone. - AUTODEP_OWN_FLAGS names the three tests/ rules that do not compile with $(CFLAGS) (a GPU compiler, and two benchmarks pinned to their own flags whose sources include no local header), so they write no .d. - xdna_physical_probe compiled backend_xdna.c beside its own source; it now links tests/backend_xdna_lane.o, which has exactly its flags. - clean.py removes build/segment/ as a whole, as it does build/ownership/. It used to leave the segment objects, and one left without its .d is a built target whose headers nothing tracks. - test_makefile_deps.py resolves rules spelled through variables ($(SEGMENT_BUILD_DIR)/glm.o:, $(RANSLIB):, the V4 static pattern rule) by asking make for them, so the hand-list and coverage checks see those rules instead of skipping them. The run targets fuzz-rans, dsv4-cuda-test and dsv4-cuda-loader-test keep their lists: their names are not files, so they rebuild every time. --- c/Makefile | 103 +++++++++++++++------------------- c/tests/test_makefile_deps.py | 93 ++++++++++++++++++++---------- c/tools/clean.py | 12 ++-- 3 files changed, 116 insertions(+), 92 deletions(-) diff --git a/c/Makefile b/c/Makefile index 3c2404939..058c46432 100644 --- a/c/Makefile +++ b/c/Makefile @@ -1024,7 +1024,7 @@ else RANSLIB = tools/librans_c.so endif rans: $(RANSLIB) -$(RANSLIB): tools/rans_ctypes.c rans.h +$(RANSLIB): tools/rans_ctypes.c $(CC) $(CFLAGS) -fPIC -shared $< -o $@ $(LDFLAGS) cuda-test: tests/test_mxfp4_expert_cuda.cu backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu tests/test_cuda_init_failure.cu @@ -1312,7 +1312,8 @@ tests/test_qwen36_tokenizer$(EXE): tests/test_qwen36_tokenizer.c qwen36.c $(QWEN $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # Reproducible local timing evidence; intentionally not a noisy CI perf gate. -tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_OBJ) $(CUDA_OBJ) +tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c qwen36.c \ + $(QWEN36_TIER_OBJ) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_OBJ) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) inkling$(EXE): inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) @@ -1414,8 +1415,8 @@ tests/test_xdna_failure$(EXE): tests/test_xdna_failure.c colibri.c tests/backend # # M-list defaults to 1,32,64 (the M64 bucket); pass 65,130,256 to qualify # the M256 bucket through the same owner. -tests/xdna_physical_probe$(EXE): tests/xdna_physical_probe.c colibri.c backend_xdna.c backend_xdna.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) -DCOLI_XDNA $< backend_xdna.c -o $@ $(LDFLAGS) +tests/xdna_physical_probe$(EXE): tests/xdna_physical_probe.c colibri.c tests/backend_xdna_lane.o + $(CC) $(CFLAGS) -DCOLI_XDNA $< tests/backend_xdna_lane.o -o $@ $(LDFLAGS) tests/test_json$(EXE): tests/test_json.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1513,7 +1514,7 @@ tests/test_st_f16_bf16_simd_sse41$(EXE): tests/test_st_f16_bf16_simd.c # bf16_to_f32_bulk/f16_to_f32_bulk vs the scalar per-element reference, NOT a # test gate. Build on demand: make tests/bench_st_f16_bf16_simd ARCH=native -tests/bench_st_f16_bf16_simd$(EXE): tests/bench_st_f16_bf16_simd.c st.h json.h compat.h +tests/bench_st_f16_bf16_simd$(EXE): tests/bench_st_f16_bf16_simd.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Synthetic peak-RSS proxy for qwen36.c's dense-int8-during-load change @@ -1628,7 +1629,8 @@ tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling # Reproducible local A/B for the shared-expert prefill path. This is timing # evidence, not a CI gate: build and run it explicitly on the target CPU. -tests/bench_inkling_shared_batch$(EXE): tests/bench_inkling_shared_batch.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) +tests/bench_inkling_shared_batch$(EXE): tests/bench_inkling_shared_batch.c inkling.c \ + $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) tests/test_inkling_cache_index$(EXE): tests/test_inkling_cache_index.c inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) @@ -1685,7 +1687,7 @@ tests/test_topp$(EXE): tests/test_topp.c colibri.c $(VK_OBJ) # bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test # gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp -tests/bench_topp$(EXE): tests/bench_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_topp$(EXE): tests/bench_topp.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c $(VK_OBJ) @@ -1827,58 +1829,37 @@ SEGMENT_RUNTIME_LIB = $(SEGMENT_BUILD_DIR)/libcolibri_segment_edge.a $(SEGMENT_BUILD_DIR): mkdir -p $@ -$(SEGMENT_BUILD_DIR)/glm.o: colibri.c sse41_kernels.h oracle.h segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/glm.o: colibri.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DCOLIBRI_NO_MAIN -c colibri.c -o $@ -$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c sse41_kernels.h segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h st.h quant.h fp8_format.h tok.h hyper_connections.h \ - delta_attention.h sparse_index.h vision_tower.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DGLM53_NO_MAIN -c glm53.c -o $@ -$(SEGMENT_BUILD_DIR)/inkling.o: inkling.c segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/inkling.o: inkling.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DINKLING_NO_MAIN -c inkling.c -o $@ -$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c sse41_kernels.h segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DKIMI_K3_NO_MAIN -c kimi_k3.c -o $@ -$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c sse41_kernels.h segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DOLMOE_NO_MAIN -c olmoe.c -o $@ -$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c gsgemv.h qgemv.h sse41_kernels.h segment_runtime.h edge_runtime.h \ - segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN36_NO_MAIN -c qwen36.c -o $@ -$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c sse41_kernels.h qwen38_core.h kv_prefix.h segment_runtime.h \ - segment_adapters.h segment_adapter_internal.h edge_runtime.h \ - edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h \ - qwen38_nfc.h qwen38_nfc_tables.h route_trace.h tok_unicode.h \ - tok_unicode_o200k.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN38_NO_MAIN -c qwen38.c -o $@ -$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h \ - deepseek_v4_internal.h segment_runtime.h segment_adapters.h \ - edge_runtime.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -D$* -c deepseek_v4.c -o $@ -$(SEGMENT_V4_REGISTRY_OBJ): expert_store_registry.c expert_store_registry.h \ - expert_store.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_V4_REGISTRY_OBJ): expert_store_registry.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -c expert_store_registry.c -o $@ -$(SEGMENT_BUILD_DIR)/segment_runtime.o: segment_runtime.c segment_runtime.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/segment_runtime.o: segment_runtime.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -c segment_runtime.c -o $@ -$(SEGMENT_BUILD_DIR)/edge_runtime.o: edge_runtime.c edge_runtime.h | $(SEGMENT_BUILD_DIR) +$(SEGMENT_BUILD_DIR)/edge_runtime.o: edge_runtime.c | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -c edge_runtime.c -o $@ $(SEGMENT_RUNTIME_LIB): $(SEGMENT_RUNTIME_OBJS) $(SEGMENT_ALL_OBJS) @@ -2017,22 +1998,22 @@ V4_OWNERSHIP_OBJS = \ $(V4_OWN_DIR): mkdir -p $(V4_OWN_DIR) -$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_RUNTIME -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_CONFIG -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h st.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_ST -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h quant.h fp8_format.h idot.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT -c deepseek_v4.c -o $@ # The engine (RUNTIME unit) opens its expert store through the pluggable # backend registry, so any test linking RUNTIME also links the registry object # (which defines coli_expert_store_backend_open_selected + registers "auto"). -$(V4_OWN_DIR)/expert_store_registry.o: expert_store_registry.c expert_store_registry.h expert_store.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/expert_store_registry.o: expert_store_registry.c | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -c expert_store_registry.c -o $@ tests/test_v4_ownership$(EXE): tests/test_v4_ownership.c $(V4_OWNERSHIP_OBJS) @@ -2213,12 +2194,12 @@ tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c colibri.c $(VK_OBJ) # bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356), # NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select -tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_router_select is a microbenchmark (duplicate-prefix scan vs marked-score scan), # not a test gate. Build on demand: make tests/bench_router_select. -tests/bench_router_select$(EXE): tests/bench_router_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_router_select$(EXE): tests/bench_router_select.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_indexer_allocations: microbenchmark (DeepSeek V4 indexer malloc vs persistent arena scratch), NOT a test gate. @@ -2228,23 +2209,23 @@ tests/bench_indexer_allocations$(EXE): tests/bench_indexer_allocations.c # bench_idot: microbenchmark (single-acc vs independent-acc AVX-VNNI idot), NOT a test gate. # Build on demand on an AVX-VNNI CPU: make tests/bench_idot ARCH=native -tests/bench_idot$(EXE): tests/bench_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_idot$(EXE): tests/bench_idot.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_i4p_gidot: microbenchmark (per-row vs multi-row/AMX K1b grouped planar IDOT), NOT a test gate. # Build on demand: make tests/bench_i4p_gidot ARCH=native -tests/bench_i4p_gidot$(EXE): tests/bench_i4p_gidot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_i4p_gidot$(EXE): tests/bench_i4p_gidot.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_gemv_stream: microbenchmark (decode-regime GEMV bandwidth vs the read ceiling; # frozen-baseline + deinterleaved-x candidate A/B), NOT a test gate. # Build on demand: make tests/bench_gemv_stream ARCH=native -tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_mla_simd: microbenchmark (scalar vs AVX2/NEON MLA-absorb reductions, #442), # NOT a test gate. Build on demand: make tests/bench_mla_simd -tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c colibri.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_uring$(EXE): tests/test_uring.c colibri.c $(VK_OBJ) @@ -2472,8 +2453,9 @@ tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_ # # Every rule in AUTODEP_BINS compiles ONE translation unit with $(CFLAGS), so # -MMD (added to CFLAGS near the top) leaves a complete .d and the rule -# does not list headers by hand. The engines and tests are derived the same way -# TEST_RULES is, so a new rule joins automatically. +# does not list headers by hand. The engines and everything under tests/ are +# derived the same way TEST_RULES is, so a new rule joins automatically; the +# objects and the ctypes library are named here. # # A command that compiles several .c files writes a .d for the LAST one only, # silently dropping the headers of the others. So a helper source a test links @@ -2483,9 +2465,11 @@ tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_ # # AUTODEP_ELSEWHERE keeps its header lists: test_deepseek_v4 and # test_v4_serve_framing are built by Makefile.deepseek-v4, and the four -# segment/edge adapter tests still compile three units in one command. The -# nvcc, MSVC-hosted CUDA and Metal objects keep theirs too and are not listed -# here at all. +# segment/edge adapter tests still compile three units in one command. +# AUTODEP_OWN_FLAGS does not compile through $(CFLAGS) at all (a GPU compiler, +# or a benchmark pinned to its own flags), so it writes no .d. The nvcc, +# MSVC-hosted CUDA, Metal and fuzz targets, and the gcc-built CUDA loader +# harnesses (dsv4-cuda-*), keep their lists and are not listed here at all. # # Most rules do not depend on .build-config, so a change to CFLAGS alone would # not rebuild a binary made before this change, and it would have no .d. Each @@ -2495,13 +2479,18 @@ tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c kimi_k3.c $(VK_ AUTODEP_ELSEWHERE = test_segment_adapters_registration test_segment_adapters_real \ test_edge_adapters_registration test_edge_adapters_real test_deepseek_v4 \ test_v4_serve_framing +AUTODEP_OWN_FLAGS = bench_cuda_resident_batch bench_load_tq_peak_rss \ + bench_indexer_allocations ENGINE_RULES := $(foreach n,$(RULE_NAMES),$(if $(findstring /,$(n)),,$(if $(wildcard $(n).c),$(n)))) +AUTODEP_TESTS := $(patsubst tests/%,%,$(filter tests/%,$(RULE_NAMES))) AUTODEP_OBJS = backend_vulkan.o backend_loader.o backend_xdna.o qwen36_tier.o \ tests/backend_xdna_lane.o tests/edge_runtime.o tests/edge_runtime_nocuda.o \ tests/segment_runtime.o tests/segment_runtime_nocuda.o \ - tests/segment_conformance_fixtures.o tests/deepseek_v4_native_quant.o + tests/segment_conformance_fixtures.o tests/deepseek_v4_native_quant.o \ + $(SEGMENT_ALL_OBJS) $(SEGMENT_RUNTIME_OBJS) $(V4_OWNERSHIP_OBJS) $(RANSLIB) AUTODEP_BINS = $(addsuffix $(EXE),$(ENGINE_RULES)) \ - $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_ELSEWHERE),$(TEST_RULES)))) \ + $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(AUTODEP_ELSEWHERE) \ + $(AUTODEP_OWN_FLAGS),$(AUTODEP_TESTS)))) \ $(AUTODEP_OBJS) AUTODEP_DEPS = $(addsuffix .d,$(basename $(AUTODEP_BINS))) $(foreach b,$(AUTODEP_BINS),$(eval $(b): $(basename $(b)).d)) diff --git a/c/tests/test_makefile_deps.py b/c/tests/test_makefile_deps.py index 741f3a457..835b43015 100644 --- a/c/tests/test_makefile_deps.py +++ b/c/tests/test_makefile_deps.py @@ -110,18 +110,58 @@ def _make(*args): capture_output=True, timeout=600) -def _make_var(name): +def _make_vars(*names): + """Makefile variables as make expands them, {name: words}, in one call.""" # `make --eval` needs GNU Make 3.82; macOS ships 3.81, which does read an # extra makefile from stdin. with _build_config_kept(): proc = subprocess.run([MAKE, "-s", "--no-print-directory", "-f", "Makefile", - "-f", "-", f"print-{name}"], cwd=C_DIR, text=True, - input="print-%: ; @echo $($*)\n", capture_output=True, - timeout=300) + "-f", "-", *(f"print-{n}" for n in names)], cwd=C_DIR, + text=True, input="print-%: ; @echo $*=$($*)\n", + capture_output=True, timeout=300) if proc.returncode != 0: - raise AssertionError(f"could not read {name} from the Makefile:\n" + raise AssertionError(f"could not read {' '.join(names)} from the Makefile:\n" f"{proc.stderr}") - return proc.stdout.split() + values = {} + for line in proc.stdout.splitlines(): + name, sep, value = line.partition("=") + if sep and name in names: + values[name] = value.split() + return values + + +def _make_var(name): + return _make_vars(name).get(name, []) + + +RULE_LINE_RE = re.compile(r"(?m)^([^\s#:=][^:=\n]*?):(?![=:])(.*)$") +VAR_RE = re.compile(r"\$\(([A-Za-z0-9_]+)\)") + + +def _rules_by_target(): + """Each rule's prerequisites as written, keyed by the file make builds. + + Rules are spelled `colibri$(EXE):`, `$(SEGMENT_BUILD_DIR)/glm.o:`, + `$(RANSLIB):` or `$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: ...`, so + the target side is expanded by asking make for those variables, and a + static pattern rule gives its prerequisites to each object it names. The + prerequisites keep their spelling: a hand-listed header is a plain word. + """ + # `$(foreach b,...,$(eval ...))` lines generate rules; they are not rules. + lines = [(t, r) for t, r in RULE_LINE_RE.findall(_joined_makefile()) + if not re.search(r"\$\([a-z-]+ ", t)] + names = sorted({v for target, _ in lines for v in VAR_RE.findall(target)}) + values = _make_vars(*names) + rules = {} + for target, rest in lines: + prereqs = rest.split("#", 1)[0] + pattern, sep, after = prereqs.partition(":") + if sep and "%" in pattern: + prereqs = after + expanded = VAR_RE.sub(lambda m: " ".join(values.get(m.group(1), [])), target) + for name in expanded.split(): + rules.setdefault(name, []).extend(prereqs.split()) + return rules def _commands(targets, *variables): @@ -208,21 +248,11 @@ def _unconditional_includes(source): return found -def _rule_form(target, exe): - """How a built target is spelled in the Makefile: colibri.exe -> colibri$(EXE).""" - if target.endswith(".o"): - return target - if exe and target.endswith(exe): - return target[:-len(exe)] + "$(EXE)" - return target + "$(EXE)" - - -def _source_of(form, rules_text): +def _source_of(target, rules): """The .c a rule compiles: the first source among its prerequisites.""" - for m in re.finditer(r"(?m)^" + re.escape(form) + r":[ \t]*(.*)$", rules_text): - for word in m.group(1).split("#", 1)[0].split(): - if SOURCE_RE.search(word): - return C_DIR / word + for word in rules.get(target, []): + if SOURCE_RE.search(word): + return C_DIR / word return None @@ -266,16 +296,19 @@ def test_every_engine_gets_generated_deps(self): def test_no_generated_rule_lists_headers_by_hand(self): """A hand-listed header on these rules is redundant and brings back the - merge conflicts #1741 removed. Bite: add `st.h` to `colibri$(EXE):`.""" - text = _joined_makefile() + merge conflicts #1741 removed. Bite: add `st.h` to `colibri$(EXE):`, + or to `$(SEGMENT_BUILD_DIR)/glm.o:`.""" + rules = _rules_by_target() problems = [] for target in self.bins: - form = _rule_form(target, self.exe) - for m in re.finditer(r"(?m)^" + re.escape(form) + r":[ \t]*(.*)$", text): - listed = [w for w in m.group(1).split("#", 1)[0].split() - if re.search(r"\.(h|inc)$", w) and not w.startswith("$")] - if listed: - problems.append(f"{form}: {' '.join(listed)}") + if target not in rules: + problems.append(f"{target}: no rule found for it, so its prerequisites " + f"cannot be checked") + continue + listed = [w for w in rules[target] + if re.search(r"\.(h|inc)$", w) and not w.startswith("$")] + if listed: + problems.append(f"{target}: {' '.join(listed)}") self.assertEqual(problems, [], "these rules get their headers from -MMD and should " "not list them by hand:\n " + "\n ".join(problems)) @@ -345,7 +378,7 @@ def setUpClass(cls): "this after `make check`") def test_every_built_target_has_a_dep_file_covering_its_includes(self): - text = _joined_makefile() + rules = _rules_by_target() problems = [] for target in self.built: dep = _dep_file(target) @@ -353,7 +386,7 @@ def test_every_built_target_has_a_dep_file_covering_its_includes(self): problems.append(f"{target}: built, but has no {dep.name}; its " f"headers are untracked until it is rebuilt") continue - source = _source_of(_rule_form(target, self.exe), text) + source = _source_of(target, rules) if source is None: problems.append(f"{target}: no source found in its rule") continue diff --git a/c/tools/clean.py b/c/tools/clean.py index 6805529cd..33c005a4b 100644 --- a/c/tools/clean.py +++ b/c/tools/clean.py @@ -67,17 +67,19 @@ # (#1741). A stale one only adds prerequisites, but clean should leave nothing # the build made, and removing it forces the rebuild that writes a fresh one. # They land wherever an output does: c/ and tests/ for the engines and tests, -# tools/ for the ctypes libraries, build/segment/ for the V4 unit objects -# (build/ownership/ goes as a whole directory below). +# tools/ for the ctypes library; build/segment/ and build/ownership/ go as +# whole directories below. ARTIFACT_GLOBS = ["tests/test_*", "tests/bench_*", "tests/fuzz_*", "tests/*_probe*", "COLI_V4_UNIT_*.o", "*.d", "tests/*.d", - "tools/*.d", "build/segment/*.d", + "tools/*.d", # helper objects the tests link (#1741), one unit per command "tests/*.o"] KEEP_EXT = (".c", ".h", ".cc", ".cpp", ".cu", ".mm", ".py", ".txt", ".json", ".md", ".bin", ".sh", ".toml", ".yml", ".yaml") -# Directories to remove. -DIRS = ["tests/__pycache__", "build/ownership"] +# Directories to remove. build/segment/ holds only build output (the +# segment-library objects, their .d and the archive); an object left there +# without its .d would be a built target whose headers are untracked (#1741). +DIRS = ["tests/__pycache__", "build/ownership", "build/segment"] def clean(): """Remove everything above, relative to the current directory."""