diff --git a/c/Makefile b/c/Makefile index 21c5988fb..786bf4b70 100644 --- a/c/Makefile +++ b/c/Makefile @@ -2630,3 +2630,11 @@ AUTODEP_DEPS = $(addsuffix .d,$(basename $(AUTODEP_BINS))) $(foreach b,$(AUTODEP_BINS),$(eval $(b): $(basename $(b)).d)) $(AUTODEP_DEPS): ; -include $(wildcard $(AUTODEP_DEPS)) + +# Read side of the same class: buffer sized from the tensor, read with config dims. +tests/test_colibri_read_trust$(EXE): tests/test_colibri_read_trust.c tests/st_fixture.h colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_inkling_read_trust$(EXE): tests/test_inkling_read_trust.c tests/st_fixture.h inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) + $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) +tests/test_olmoe_read_trust$(EXE): tests/test_olmoe_read_trust.c tests/st_fixture.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h + $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) diff --git a/c/colibri.c b/c/colibri.c index 20b9048de..11b5ae04a 100644 --- a/c/colibri.c +++ b/c/colibri.c @@ -2304,10 +2304,15 @@ static QT qt_load_ex(Model *m, const char *name, int O, int I, int bits, int mma return qt_load(m,name,O,I,bits); } -static float *ld(Model *m, const char *name){ /* tensore 1D f32 residente (norme/bias) */ +/* `want` is the element count the forward pass indexes with (config dims). A + * shorter tensor used to load into a buffer of its own size and be read past its + * end at inference; refuse it here, as qwen36's load_t_n does. */ +static float *ld(Model *m, const char *name, int64_t want){ /* tensore 1D f32 residente (norme/bias) */ int64_t n=st_numel(&m->S,name); if(n<0) st_die_missing(&m->S,name); + if(n!=want){ fprintf(stderr,"%s: %lld elements, config implies %lld -- refusing\n", + name,(long long)n,(long long)want); exit(1); } float *p=(float*)qalloc((size_t)n*sizeof(float)); /* registrato per la GPU sotto METAL */ - st_read_f32(&m->S,name,p,0); return p; + st_read_f32_cap(&m->S,name,p,want,0); return p; } #ifdef COLI_CUDA static void qt_cuda_colocate(QT *dst,const QT *src){ @@ -2483,7 +2488,7 @@ static void model_init_range(Model *m, const char *snap, int cap, if(load_boundaries){ m->embed = qt_load(m,"model.embed_tokens.weight", c->vocab, D, io_bits); m->lm_head = qt_load(m,"lm_head.weight", c->vocab, D, io_bits); - m->final_norm = ld(m,"model.norm.weight"); + m->final_norm = ld(m,"model.norm.weight",D); } m->L=calloc(c->n_layers,sizeof(Layer)); int NR=c->n_layers+1; /* +1: riga del layer MTP */ @@ -2533,13 +2538,13 @@ static void model_init_range(Model *m, const char *snap, int cap, && (i < c->n_layers - g_trunk_resident); l->trunk_mmap = mmap_ok; #define P(s) (snprintf(nm,sizeof(nm),"model.layers.%d." s,i),nm) - l->in_ln=ld(m,P("input_layernorm.weight")); - l->post_ln=ld(m,P("post_attention_layernorm.weight")); + l->in_ln=ld(m,P("input_layernorm.weight"),D); + l->post_ln=ld(m,P("post_attention_layernorm.weight"),D); l->q_a = qt_load_ex(m,P("self_attn.q_a_proj.weight"), c->q_lora, D, dbits,mmap_ok); - l->q_a_ln= ld(m,P("self_attn.q_a_layernorm.weight")); + l->q_a_ln= ld(m,P("self_attn.q_a_layernorm.weight"),c->q_lora); l->q_b = qt_load_ex(m,P("self_attn.q_b_proj.weight"), H*c->qk_head, c->q_lora, dbits,mmap_ok); l->kv_a = qt_load_ex(m,P("self_attn.kv_a_proj_with_mqa.weight"), c->kv_lora+c->qk_rope, D, dbits,mmap_ok); - l->kv_a_ln= ld(m,P("self_attn.kv_a_layernorm.weight")); + l->kv_a_ln= ld(m,P("self_attn.kv_a_layernorm.weight"),c->kv_lora); l->kv_b = qt_load_ex(m,P("self_attn.kv_b_proj.weight"), H*(c->qk_nope+c->v_head), c->kv_lora, dbits,mmap_ok); l->o = qt_load_ex(m,P("self_attn.o_proj.weight"), D, H*c->v_head, dbits,mmap_ok); #ifdef COLI_CUDA @@ -2559,8 +2564,8 @@ static void model_init_range(Model *m, const char *snap, int cap, if(!l->up_proj.mmap_view) qt_planarize(&l->up_proj); if(!l->down_proj.mmap_view) qt_planarize(&l->down_proj); } else { - l->router=ld(m,P("mlp.gate.weight")); - l->router_bias=ld(m,P("mlp.gate.e_score_correction_bias")); + l->router=ld(m,P("mlp.gate.weight"),(int64_t)c->n_experts*D); + l->router_bias=ld(m,P("mlp.gate.e_score_correction_bias"),c->n_experts); int sI=c->moe_inter*c->n_shared; l->sh_gate = qt_load_ex(m,P("mlp.shared_experts.gate_proj.weight"), sI, D, dbits,mmap_ok); l->sh_up = qt_load_ex(m,P("mlp.shared_experts.up_proj.weight"), sI, D, dbits,mmap_ok); @@ -2607,25 +2612,25 @@ static void model_init_range(Model *m, const char *snap, int cap, if(m->has_mtp){ int i=c->n_layers; Layer *l=&m->mtpL; #define PM(s) (snprintf(nm,sizeof(nm),"model.layers.%d." s,i),nm) - l->in_ln=ld(m,PM("input_layernorm.weight")); - l->post_ln=ld(m,PM("post_attention_layernorm.weight")); + l->in_ln=ld(m,PM("input_layernorm.weight"),D); + l->post_ln=ld(m,PM("post_attention_layernorm.weight"),D); l->q_a = qt_load(m,PM("self_attn.q_a_proj.weight"), c->q_lora, D, dbits); - l->q_a_ln= ld(m,PM("self_attn.q_a_layernorm.weight")); + l->q_a_ln= ld(m,PM("self_attn.q_a_layernorm.weight"),c->q_lora); l->q_b = qt_load(m,PM("self_attn.q_b_proj.weight"), H*c->qk_head, c->q_lora, dbits); l->kv_a = qt_load(m,PM("self_attn.kv_a_proj_with_mqa.weight"), c->kv_lora+c->qk_rope, D, dbits); - l->kv_a_ln= ld(m,PM("self_attn.kv_a_layernorm.weight")); + l->kv_a_ln= ld(m,PM("self_attn.kv_a_layernorm.weight"),c->kv_lora); l->kv_b = qt_load(m,PM("self_attn.kv_b_proj.weight"), H*(c->qk_nope+c->v_head), c->kv_lora, dbits); l->o = qt_load(m,PM("self_attn.o_proj.weight"), D, H*c->v_head, dbits); l->sparse=1; - l->router=ld(m,PM("mlp.gate.weight")); - l->router_bias=ld(m,PM("mlp.gate.e_score_correction_bias")); + l->router=ld(m,PM("mlp.gate.weight"),(int64_t)c->n_experts*D); + l->router_bias=ld(m,PM("mlp.gate.e_score_correction_bias"),c->n_experts); int sI=c->moe_inter*c->n_shared; l->sh_gate = qt_load(m,PM("mlp.shared_experts.gate_proj.weight"), sI, D, dbits); l->sh_up = qt_load(m,PM("mlp.shared_experts.up_proj.weight"), sI, D, dbits); l->sh_down = qt_load(m,PM("mlp.shared_experts.down_proj.weight"), D, sI, dbits); m->eh_proj = qt_load(m,PM("eh_proj.weight"), D, 2*D, dbits); - m->enorm=ld(m,PM("enorm.weight")); m->hnorm=ld(m,PM("hnorm.weight")); - m->mtp_norm=ld(m,PM("shared_head.norm.weight")); + m->enorm=ld(m,PM("enorm.weight"),D); m->hnorm=ld(m,PM("hnorm.weight"),D); + m->mtp_norm=ld(m,PM("shared_head.norm.weight"),D); m->ecache[i]=calloc(cap,sizeof(ESlot)); m->eroute[i]=calloc(c->topk,sizeof(int)); m->eheat[i]=calloc(c->n_experts,sizeof(uint32_t)); @@ -2657,7 +2662,7 @@ static void model_init_range(Model *m, const char *snap, int cap, m->ix_wq[i]=qt_load(m,PI("wq_b.weight"), c->index_nh*c->index_hd, c->q_lora, dbits); m->ix_wk[i]=qt_load(m,PI("wk.weight"), c->index_hd, D, dbits); m->ix_wp[i]=qt_load(m,PI("weights_proj.weight"), c->index_nh, D, dbits); - m->ix_knw[i]=ld(m,PI("k_norm.weight")); m->ix_knb[i]=ld(m,PI("k_norm.bias")); + m->ix_knw[i]=ld(m,PI("k_norm.weight"),c->index_hd); m->ix_knb[i]=ld(m,PI("k_norm.bias"),c->index_hd); #undef PI } fprintf(stderr,"[DSA] indexer active: top-%d sparse attention beyond %d context tokens\n", @@ -13492,7 +13497,7 @@ static int glm_edge_engine_open( model->c.vocab, model->c.hidden, io_bits); model->lm_head = qt_load(model, "lm_head.weight", model->c.vocab, model->c.hidden, io_bits); - model->final_norm = ld(model, "model.norm.weight"); + model->final_norm = ld(model, "model.norm.weight", model->c.hidden); char tokenizer_path[4096]; snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json", options->model_dir); diff --git a/c/inkling.c b/c/inkling.c index cfb519337..b62b8038b 100644 --- a/c/inkling.c +++ b/c/inkling.c @@ -718,11 +718,17 @@ static const char *embed_norm_name(shards *S) { return NULL; } -static float *load_t(Model *m, const char *name) { +/* `want` is the element count the forward pass indexes with (config dims); a + * short tensor used to be read past its end at inference (see qwen36 load_t_n). */ +static float *load_t(Model *m, const char *name, int64_t want) { int64_t n = st_numel(&m->S, name); if (n < 0) { fprintf(stderr, "missing %s\n", name); exit(1); } + if (n != want) { + fprintf(stderr, "%s: %lld elements, config implies %lld -- refusing\n", + name, (long long)n, (long long)want); exit(1); + } float *p = falloc(n); - st_read_f32(&m->S, name, p, 0); + st_read_f32_cap(&m->S, name, p, want, 0); return p; } static float load_scalar(Model *m, const char *name, float dflt) { @@ -951,8 +957,8 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, #endif if (load_boundaries) { m->embed = load_w(m, "model.embed_tokens.weight", 0); - { const char *en = embed_norm_name(&m->S); m->embed_norm = en ? load_t(m, en) : NULL; } - m->final_norm = load_t(m, "model.norm.weight"); + { const char *en = embed_norm_name(&m->S); m->embed_norm = en ? load_t(m, en, D) : NULL; } + m->final_norm = load_t(m, "model.norm.weight", D); m->lm_head = load_w(m, "lm_head.weight", 1); } /* Inkling's audio "tower" is one embedding table + one RMSNorm. The int4 @@ -976,7 +982,7 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, exit(1); } m->audio_enc = load_w(m, "model.audio.encoder.weight", 0); - m->audio_norm = load_t(m, "model.audio.final_norm.weight"); + m->audio_norm = load_t(m, "model.audio.final_norm.weight", D); fprintf(stderr, "[audio] DMel encoder loaded (%d bins x %d levels -> D=%d)\n", c->mel_bins, c->mel_vocab, D); } @@ -986,25 +992,26 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, char nm[320]; for (int i = layer_begin; i < layer_end; i++) { Layer *l = &m->L[i]; - #define LD(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm) + #define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm,want) + int64_t hd = L_HD(c,i), kvd = (int64_t)L_KV(c,i) * hd; #define LDW(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_w(m,nm,1) - LD(in_ln, "input_layernorm.weight"); - LD(post_ln,"post_attention_layernorm.weight"); + LD(in_ln, "input_layernorm.weight", D); + LD(post_ln,"post_attention_layernorm.weight", D); LDW(q, "self_attn.q_proj.weight"); LDW(k, "self_attn.k_proj.weight"); LDW(v, "self_attn.v_proj.weight"); LDW(r, "self_attn.r_proj.weight"); LDW(o, "self_attn.o_proj.weight"); - LD(qn,"self_attn.q_norm.weight"); LD(kn,"self_attn.k_norm.weight"); - LD(relp, "self_attn.rel_logits_proj.proj"); - LD(k_cw, "self_attn.k_sconv.conv1d.weight"); - LD(v_cw, "self_attn.v_sconv.conv1d.weight"); - LD(a_cw, "attn_sconv.conv1d.weight"); - LD(m_cw, "mlp_sconv.conv1d.weight"); + LD(qn,"self_attn.q_norm.weight", hd); LD(kn,"self_attn.k_norm.weight", hd); + LD(relp, "self_attn.rel_logits_proj.proj", (int64_t)c->d_rel * L_EXT(c,i)); + LD(k_cw, "self_attn.k_sconv.conv1d.weight", kvd * K); + LD(v_cw, "self_attn.v_sconv.conv1d.weight", kvd * K); + LD(a_cw, "attn_sconv.conv1d.weight", (int64_t)D * K); + LD(m_cw, "mlp_sconv.conv1d.weight", (int64_t)D * K); if (!c->sparse[i]) { LDW(dg, "mlp.gate_proj.weight"); LDW(du, "mlp.up_proj.weight"); LDW(dd, "mlp.down_proj.weight"); snprintf(nm,sizeof(nm),"model.layers.%d.mlp.global_scale",i); l->dgs = load_scalar(m,nm,1.f); } else { - LD(router, "mlp.gate.weight"); - LD(rbias, "mlp.gate.e_score_correction_bias"); + LD(router, "mlp.gate.weight", (int64_t)(c->n_experts + c->n_shared) * D); + LD(rbias, "mlp.gate.e_score_correction_bias", c->n_experts); snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.global_scale",i); l->rgs = load_scalar(m,nm,1.f); LDW(sh_g, "mlp.shared_experts.gate_proj"); LDW(sh_u, "mlp.shared_experts.up_proj"); @@ -3269,8 +3276,8 @@ static int inkling_edge_engine_open( } model->embed = load_w(model, "model.embed_tokens.weight", 0); const char *embed_norm = embed_norm_name(&model->S); - model->embed_norm = embed_norm ? load_t(model, embed_norm) : NULL; - model->final_norm = load_t(model, "model.norm.weight"); + model->embed_norm = embed_norm ? load_t(model, embed_norm, model->c.hidden) : NULL; + model->final_norm = load_t(model, "model.norm.weight", model->c.hidden); model->lm_head = load_w(model, "lm_head.weight", 1); char tokenizer_path[4096]; snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json", diff --git a/c/olmoe.c b/c/olmoe.c index 5afa346f1..355024e25 100644 --- a/c/olmoe.c +++ b/c/olmoe.c @@ -482,11 +482,17 @@ static void load_cfg(Cfg *c, const char *snap) { free(buf); free(arena); } -static float *load_t(Model *m, const char *name) { +/* `want` is the element count the forward pass indexes with (config dims); a + * short tensor used to be read past its end at inference (see qwen36 load_t_n). */ +static float *load_t(Model *m, const char *name, int64_t want) { int64_t n = st_numel(&m->S, name); if (n < 0) { fprintf(stderr, "missing %s\n", name); exit(1); } + if (n != want) { + fprintf(stderr, "%s: %lld elements, config implies %lld -- refusing\n", + name, (long long)n, (long long)want); exit(1); + } float *p = falloc(n); - st_read_f32(&m->S, name, p, 0); /* densa: niente DONTNEED, resta residente */ + st_read_f32_cap(&m->S, name, p, want, 0); /* densa: niente DONTNEED, resta residente */ return p; } @@ -507,21 +513,22 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, } double t0 = now_s(); if (load_boundaries) { - m->embed = load_t(m, "model.embed_tokens.weight"); - m->lm_head = load_t(m, "lm_head.weight"); - m->final_norm = load_t(m, "model.norm.weight"); + m->embed = load_t(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden); + m->lm_head = load_t(m, "lm_head.weight", (int64_t)c->vocab * c->hidden); + m->final_norm = load_t(m, "model.norm.weight", c->hidden); } m->L = calloc(c->n_layers, sizeof(Layer)); char nm[256]; for (int i = layer_begin; i < layer_end; i++) { Layer *l = &m->L[i]; - #define LD(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm) - LD(in_ln, "input_layernorm.weight"); - LD(post_ln,"post_attention_layernorm.weight"); - LD(q, "self_attn.q_proj.weight"); LD(k, "self_attn.k_proj.weight"); - LD(v, "self_attn.v_proj.weight"); LD(o, "self_attn.o_proj.weight"); - LD(qn,"self_attn.q_norm.weight"); LD(kn,"self_attn.k_norm.weight"); - LD(gate, "mlp.gate.weight"); + #define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm,want) + int64_t D = c->hidden; + LD(in_ln, "input_layernorm.weight", D); + LD(post_ln,"post_attention_layernorm.weight", D); + LD(q, "self_attn.q_proj.weight", D*D); LD(k, "self_attn.k_proj.weight", D*D); + LD(v, "self_attn.v_proj.weight", D*D); LD(o, "self_attn.o_proj.weight", D*D); + LD(qn,"self_attn.q_norm.weight", D); LD(kn,"self_attn.k_norm.weight", D); + LD(gate, "mlp.gate.weight", (int64_t)c->n_experts * D); #undef LD } /* cap <= 0 is "you decide", the sentinel the launcher sends when nobody @@ -2416,9 +2423,11 @@ static int olmoe_edge_engine_open( "out of memory opening OLMoE Edge"); load_cfg(&engine->model.c, options->model_dir); st_init(&engine->model.S, options->model_dir); - engine->model.embed = load_t(&engine->model, "model.embed_tokens.weight"); - engine->model.lm_head = load_t(&engine->model, "lm_head.weight"); - engine->model.final_norm = load_t(&engine->model, "model.norm.weight"); + int64_t vd = (int64_t)engine->model.c.vocab * engine->model.c.hidden; + engine->model.embed = load_t(&engine->model, "model.embed_tokens.weight", vd); + engine->model.lm_head = load_t(&engine->model, "lm_head.weight", vd); + engine->model.final_norm = load_t(&engine->model, "model.norm.weight", + engine->model.c.hidden); char tokenizer_path[4096]; snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json", options->model_dir); diff --git a/c/tests/st_fixture.h b/c/tests/st_fixture.h new file mode 100644 index 000000000..5ccc484ac --- /dev/null +++ b/c/tests/st_fixture.h @@ -0,0 +1,48 @@ +/* Shared by the size-trust tests: write a one-shard safetensors container whose + * tensors declare exactly the dtype, shape and byte count the case needs, then + * run each case in a child and inspect how it ended. */ +#ifndef ST_FIXTURE_H +#define ST_FIXTURE_H +#include +#include +#include +#include +#include +typedef struct { const char *name, *dtype; int64_t numel, nbytes; } FxT; +static int fx_write(const char *dir, const FxT *t, int n){ + char hdr[8192], path[512]; int len = 0; int64_t off = 0; + len += snprintf(hdr + len, sizeof hdr - len, "{"); + for (int i = 0; i < n; i++, off += t[i - 1].nbytes) + len += snprintf(hdr + len, sizeof hdr - len, + "%s\"%s\":{\"dtype\":\"%s\",\"shape\":[%lld],\"data_offsets\":[%lld,%lld]}", + i ? "," : "", t[i].name, t[i].dtype, (long long)t[i].numel, + (long long)off, (long long)(off + t[i].nbytes)); + len += snprintf(hdr + len, sizeof hdr - len, "}"); + while (len % 8) hdr[len++] = ' '; + snprintf(path, sizeof path, "%s/model.safetensors", dir); + FILE *f = fopen(path, "wb"); if (!f) return -1; + uint64_t hl = (uint64_t)len; fwrite(&hl, 8, 1, f); fwrite(hdr, 1, (size_t)len, f); + for (int64_t i = 0; i < off; i++) fputc(0x3c, f); /* 0x3c3c3c3c = 0.0115f */ + return fclose(f); +} +/* Run `self --child `; returns the exit code and fills `log` with stderr. */ +static int fx_child(const char *self, int c, const char *dir, char *log, size_t cap){ + char err[600], cmd[1600]; + snprintf(err, sizeof err, "%s/stderr.txt", dir); + snprintf(cmd, sizeof cmd, "\"%s\" --child %d \"%s\" >/dev/null 2>\"%s\"", self, c, dir, err); + int st = system(cmd), got = st >= 0 && WIFEXITED(st) ? WEXITSTATUS(st) : -1; + log[0] = 0; FILE *f = fopen(err, "rb"); + if (f) { size_t r = fread(log, 1, cap - 1, f); log[r] = 0; fclose(f); } + remove(err); + return got; +} +static void fx_cleanup(const char *dir){ + char p[600]; snprintf(p, sizeof p, "%s/model.safetensors", dir); remove(p); rmdir(dir); +} +/* A refusal is exit 1 naming the tensor, with no sanitizer report: ASan also + * exits 1, so the exit code alone cannot tell a refusal from a heap overflow. */ +static int fx_refused(int got, const char *log, const char *tensor){ + return got == 1 && strstr(log, tensor) && strstr(log, "refusing") && + !strstr(log, "Sanitizer"); +} +#endif diff --git a/c/tests/test_colibri_read_trust.c b/c/tests/test_colibri_read_trust.c new file mode 100644 index 000000000..242536fcf --- /dev/null +++ b/c/tests/test_colibri_read_trust.c @@ -0,0 +1,39 @@ +/* colibri's ld() sizes its buffer from the tensor's declared count, while the forward pass + * reads it with CONFIG dims. A tensor shorter than the config says loaded + * cleanly and was read past its end at the first rmsnorm() (run under ASan to see + * the overflow). Case 0 is a right-sized tensor and must load and be consumed + * cleanly; case 1 is one value where the config implies D and must be refused + * by name at load. */ +#define main coli_glm_main_unused +#include "../colibri.c" +#undef main +#include "st_fixture.h" +enum { D = 64 }; +static int child(int c, const char *dir){ + FxT t[1] = { {"w", "F32", c ? 1 : D, (c ? 1 : D) * 4} }; + if (fx_write(dir, t, 1)) return 90; + static Model m; memset(&m, 0, sizeof m); + st_init(&m.S, dir); + float *w = ld(&m, "w", D), x[D], out[D]; + for (int i = 0; i < D; i++) x[i] = 1.f; + rmsnorm(out, x, w, D, 1e-6f); /* the forward pass's own consumer, at config D */ + fprintf(stderr, "loaded; out[0] %g\n", out[0]); + free(w); + return 0; +} +int main(int argc, char **argv){ + if (argc == 4 && !strcmp(argv[1], "--child")) return child(atoi(argv[2]), argv[3]); + int fails = 0; + for (int c = 0; c < 2; c++) { + char dir[] = "test_colibri_read_trust_XXXXXX", log[16384]; + if (!mkdtemp(dir)) { perror("mkdtemp"); return 1; } + int got = fx_child(argv[0], c, dir, log, sizeof log); + int ok = c ? fx_refused(got, log, "w") : got == 0 && !strstr(log, "Sanitizer"); + if (!ok) { fails++; printf("FAIL: case %d: exit %d, want %s\n--- child stderr ---\n%s\n", + c, got, c ? "a refusal naming w" : "a clean load", log); } + fx_cleanup(dir); + } + if (fails) { printf("test_colibri_read_trust: %d failure(s)\n", fails); return 1; } + printf("OK test_colibri_read_trust: right-sized tensor loads, short tensor refused before a config-sized read\n"); + return 0; +} diff --git a/c/tests/test_inkling_read_trust.c b/c/tests/test_inkling_read_trust.c new file mode 100644 index 000000000..73ebf65f1 --- /dev/null +++ b/c/tests/test_inkling_read_trust.c @@ -0,0 +1,39 @@ +/* inkling's load_t() sizes its buffer from the tensor's declared count, while the forward pass + * reads it with CONFIG dims. A tensor shorter than the config says loaded + * cleanly and was read past its end at the first rmsnorm_row() (run under ASan to see + * the overflow). Case 0 is a right-sized tensor and must load and be consumed + * cleanly; case 1 is one value where the config implies D and must be refused + * by name at load. */ +#define main inkling_main_unused +#include "../inkling.c" +#undef main +#include "st_fixture.h" +enum { D = 64 }; +static int child(int c, const char *dir){ + FxT t[1] = { {"w", "F32", c ? 1 : D, (c ? 1 : D) * 4} }; + if (fx_write(dir, t, 1)) return 90; + static Model m; memset(&m, 0, sizeof m); + st_init(&m.S, dir); + float *w = load_t(&m, "w", D), x[D], out[D]; + for (int i = 0; i < D; i++) x[i] = 1.f; + rmsnorm_row(out, x, w, D, 1e-6f); /* the forward pass's own consumer, at config D */ + fprintf(stderr, "loaded; out[0] %g\n", out[0]); + free(w); + return 0; +} +int main(int argc, char **argv){ + if (argc == 4 && !strcmp(argv[1], "--child")) return child(atoi(argv[2]), argv[3]); + int fails = 0; + for (int c = 0; c < 2; c++) { + char dir[] = "test_inkling_read_trust_XXXXXX", log[16384]; + if (!mkdtemp(dir)) { perror("mkdtemp"); return 1; } + int got = fx_child(argv[0], c, dir, log, sizeof log); + int ok = c ? fx_refused(got, log, "w") : got == 0 && !strstr(log, "Sanitizer"); + if (!ok) { fails++; printf("FAIL: case %d: exit %d, want %s\n--- child stderr ---\n%s\n", + c, got, c ? "a refusal naming w" : "a clean load", log); } + fx_cleanup(dir); + } + if (fails) { printf("test_inkling_read_trust: %d failure(s)\n", fails); return 1; } + printf("OK test_inkling_read_trust: right-sized tensor loads, short tensor refused before a config-sized read\n"); + return 0; +} diff --git a/c/tests/test_olmoe_read_trust.c b/c/tests/test_olmoe_read_trust.c new file mode 100644 index 000000000..b9bd5a29a --- /dev/null +++ b/c/tests/test_olmoe_read_trust.c @@ -0,0 +1,39 @@ +/* olmoe's load_t() sizes its buffer from the tensor's declared count, while the forward pass + * reads it with CONFIG dims. A tensor shorter than the config says loaded + * cleanly and was read past its end at the first rmsnorm_row() (run under ASan to see + * the overflow). Case 0 is a right-sized tensor and must load and be consumed + * cleanly; case 1 is one value where the config implies D and must be refused + * by name at load. */ +#define main olmoe_main_unused +#include "../olmoe.c" +#undef main +#include "st_fixture.h" +enum { D = 64 }; +static int child(int c, const char *dir){ + FxT t[1] = { {"w", "F32", c ? 1 : D, (c ? 1 : D) * 4} }; + if (fx_write(dir, t, 1)) return 90; + static Model m; memset(&m, 0, sizeof m); + st_init(&m.S, dir); + float *w = load_t(&m, "w", D), x[D], out[D]; + for (int i = 0; i < D; i++) x[i] = 1.f; + rmsnorm_row(out, x, w, D, 1e-6f); /* the forward pass's own consumer, at config D */ + fprintf(stderr, "loaded; out[0] %g\n", out[0]); + free(w); + return 0; +} +int main(int argc, char **argv){ + if (argc == 4 && !strcmp(argv[1], "--child")) return child(atoi(argv[2]), argv[3]); + int fails = 0; + for (int c = 0; c < 2; c++) { + char dir[] = "test_olmoe_read_trust_XXXXXX", log[16384]; + if (!mkdtemp(dir)) { perror("mkdtemp"); return 1; } + int got = fx_child(argv[0], c, dir, log, sizeof log); + int ok = c ? fx_refused(got, log, "w") : got == 0 && !strstr(log, "Sanitizer"); + if (!ok) { fails++; printf("FAIL: case %d: exit %d, want %s\n--- child stderr ---\n%s\n", + c, got, c ? "a refusal naming w" : "a clean load", log); } + fx_cleanup(dir); + } + if (fails) { printf("test_olmoe_read_trust: %d failure(s)\n", fails); return 1; } + printf("OK test_olmoe_read_trust: right-sized tensor loads, short tensor refused before a config-sized read\n"); + return 0; +}