Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions c/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -2630,3 +2630,11 @@ AUTODEP_DEPS = $(addsuffix .d,$(basename $(AUTODEP_BINS)))
$(foreach b,$(AUTODEP_BINS),$(eval $(b): $(basename $(b)).d))
$(AUTODEP_DEPS): ;
-include $(wildcard $(AUTODEP_DEPS))

# Read side of the same class: buffer sized from the tensor, read with config dims.
tests/test_colibri_read_trust$(EXE): tests/test_colibri_read_trust.c tests/st_fixture.h colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
tests/test_inkling_read_trust$(EXE): tests/test_inkling_read_trust.c tests/st_fixture.h inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ)
$(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS)
tests/test_olmoe_read_trust$(EXE): tests/test_olmoe_read_trust.c tests/st_fixture.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
43 changes: 24 additions & 19 deletions c/colibri.c
Original file line number Diff line number Diff line change
Expand Up @@ -2304,10 +2304,15 @@ static QT qt_load_ex(Model *m, const char *name, int O, int I, int bits, int mma
return qt_load(m,name,O,I,bits);
}

static float *ld(Model *m, const char *name){ /* tensore 1D f32 residente (norme/bias) */
/* `want` is the element count the forward pass indexes with (config dims). A
* shorter tensor used to load into a buffer of its own size and be read past its
* end at inference; refuse it here, as qwen36's load_t_n does. */
static float *ld(Model *m, const char *name, int64_t want){ /* tensore 1D f32 residente (norme/bias) */
int64_t n=st_numel(&m->S,name); if(n<0) st_die_missing(&m->S,name);
if(n!=want){ fprintf(stderr,"%s: %lld elements, config implies %lld -- refusing\n",
name,(long long)n,(long long)want); exit(1); }
float *p=(float*)qalloc((size_t)n*sizeof(float)); /* registrato per la GPU sotto METAL */
st_read_f32(&m->S,name,p,0); return p;
st_read_f32_cap(&m->S,name,p,want,0); return p;
}
#ifdef COLI_CUDA
static void qt_cuda_colocate(QT *dst,const QT *src){
Expand Down Expand Up @@ -2483,7 +2488,7 @@ static void model_init_range(Model *m, const char *snap, int cap,
if(load_boundaries){
m->embed = qt_load(m,"model.embed_tokens.weight", c->vocab, D, io_bits);
m->lm_head = qt_load(m,"lm_head.weight", c->vocab, D, io_bits);
m->final_norm = ld(m,"model.norm.weight");
m->final_norm = ld(m,"model.norm.weight",D);
}
m->L=calloc(c->n_layers,sizeof(Layer));
int NR=c->n_layers+1; /* +1: riga del layer MTP */
Expand Down Expand Up @@ -2533,13 +2538,13 @@ static void model_init_range(Model *m, const char *snap, int cap,
&& (i < c->n_layers - g_trunk_resident);
l->trunk_mmap = mmap_ok;
#define P(s) (snprintf(nm,sizeof(nm),"model.layers.%d." s,i),nm)
l->in_ln=ld(m,P("input_layernorm.weight"));
l->post_ln=ld(m,P("post_attention_layernorm.weight"));
l->in_ln=ld(m,P("input_layernorm.weight"),D);
l->post_ln=ld(m,P("post_attention_layernorm.weight"),D);
l->q_a = qt_load_ex(m,P("self_attn.q_a_proj.weight"), c->q_lora, D, dbits,mmap_ok);
l->q_a_ln= ld(m,P("self_attn.q_a_layernorm.weight"));
l->q_a_ln= ld(m,P("self_attn.q_a_layernorm.weight"),c->q_lora);
l->q_b = qt_load_ex(m,P("self_attn.q_b_proj.weight"), H*c->qk_head, c->q_lora, dbits,mmap_ok);
l->kv_a = qt_load_ex(m,P("self_attn.kv_a_proj_with_mqa.weight"), c->kv_lora+c->qk_rope, D, dbits,mmap_ok);
l->kv_a_ln= ld(m,P("self_attn.kv_a_layernorm.weight"));
l->kv_a_ln= ld(m,P("self_attn.kv_a_layernorm.weight"),c->kv_lora);
l->kv_b = qt_load_ex(m,P("self_attn.kv_b_proj.weight"), H*(c->qk_nope+c->v_head), c->kv_lora, dbits,mmap_ok);
l->o = qt_load_ex(m,P("self_attn.o_proj.weight"), D, H*c->v_head, dbits,mmap_ok);
#ifdef COLI_CUDA
Expand All @@ -2559,8 +2564,8 @@ static void model_init_range(Model *m, const char *snap, int cap,
if(!l->up_proj.mmap_view) qt_planarize(&l->up_proj);
if(!l->down_proj.mmap_view) qt_planarize(&l->down_proj);
} else {
l->router=ld(m,P("mlp.gate.weight"));
l->router_bias=ld(m,P("mlp.gate.e_score_correction_bias"));
l->router=ld(m,P("mlp.gate.weight"),(int64_t)c->n_experts*D);
l->router_bias=ld(m,P("mlp.gate.e_score_correction_bias"),c->n_experts);
int sI=c->moe_inter*c->n_shared;
l->sh_gate = qt_load_ex(m,P("mlp.shared_experts.gate_proj.weight"), sI, D, dbits,mmap_ok);
l->sh_up = qt_load_ex(m,P("mlp.shared_experts.up_proj.weight"), sI, D, dbits,mmap_ok);
Expand Down Expand Up @@ -2607,25 +2612,25 @@ static void model_init_range(Model *m, const char *snap, int cap,
if(m->has_mtp){
int i=c->n_layers; Layer *l=&m->mtpL;
#define PM(s) (snprintf(nm,sizeof(nm),"model.layers.%d." s,i),nm)
l->in_ln=ld(m,PM("input_layernorm.weight"));
l->post_ln=ld(m,PM("post_attention_layernorm.weight"));
l->in_ln=ld(m,PM("input_layernorm.weight"),D);
l->post_ln=ld(m,PM("post_attention_layernorm.weight"),D);
l->q_a = qt_load(m,PM("self_attn.q_a_proj.weight"), c->q_lora, D, dbits);
l->q_a_ln= ld(m,PM("self_attn.q_a_layernorm.weight"));
l->q_a_ln= ld(m,PM("self_attn.q_a_layernorm.weight"),c->q_lora);
l->q_b = qt_load(m,PM("self_attn.q_b_proj.weight"), H*c->qk_head, c->q_lora, dbits);
l->kv_a = qt_load(m,PM("self_attn.kv_a_proj_with_mqa.weight"), c->kv_lora+c->qk_rope, D, dbits);
l->kv_a_ln= ld(m,PM("self_attn.kv_a_layernorm.weight"));
l->kv_a_ln= ld(m,PM("self_attn.kv_a_layernorm.weight"),c->kv_lora);
l->kv_b = qt_load(m,PM("self_attn.kv_b_proj.weight"), H*(c->qk_nope+c->v_head), c->kv_lora, dbits);
l->o = qt_load(m,PM("self_attn.o_proj.weight"), D, H*c->v_head, dbits);
l->sparse=1;
l->router=ld(m,PM("mlp.gate.weight"));
l->router_bias=ld(m,PM("mlp.gate.e_score_correction_bias"));
l->router=ld(m,PM("mlp.gate.weight"),(int64_t)c->n_experts*D);
l->router_bias=ld(m,PM("mlp.gate.e_score_correction_bias"),c->n_experts);
int sI=c->moe_inter*c->n_shared;
l->sh_gate = qt_load(m,PM("mlp.shared_experts.gate_proj.weight"), sI, D, dbits);
l->sh_up = qt_load(m,PM("mlp.shared_experts.up_proj.weight"), sI, D, dbits);
l->sh_down = qt_load(m,PM("mlp.shared_experts.down_proj.weight"), D, sI, dbits);
m->eh_proj = qt_load(m,PM("eh_proj.weight"), D, 2*D, dbits);
m->enorm=ld(m,PM("enorm.weight")); m->hnorm=ld(m,PM("hnorm.weight"));
m->mtp_norm=ld(m,PM("shared_head.norm.weight"));
m->enorm=ld(m,PM("enorm.weight"),D); m->hnorm=ld(m,PM("hnorm.weight"),D);
m->mtp_norm=ld(m,PM("shared_head.norm.weight"),D);
m->ecache[i]=calloc(cap,sizeof(ESlot));
m->eroute[i]=calloc(c->topk,sizeof(int));
m->eheat[i]=calloc(c->n_experts,sizeof(uint32_t));
Expand Down Expand Up @@ -2657,7 +2662,7 @@ static void model_init_range(Model *m, const char *snap, int cap,
m->ix_wq[i]=qt_load(m,PI("wq_b.weight"), c->index_nh*c->index_hd, c->q_lora, dbits);
m->ix_wk[i]=qt_load(m,PI("wk.weight"), c->index_hd, D, dbits);
m->ix_wp[i]=qt_load(m,PI("weights_proj.weight"), c->index_nh, D, dbits);
m->ix_knw[i]=ld(m,PI("k_norm.weight")); m->ix_knb[i]=ld(m,PI("k_norm.bias"));
m->ix_knw[i]=ld(m,PI("k_norm.weight"),c->index_hd); m->ix_knb[i]=ld(m,PI("k_norm.bias"),c->index_hd);
#undef PI
}
fprintf(stderr,"[DSA] indexer active: top-%d sparse attention beyond %d context tokens\n",
Expand Down Expand Up @@ -13492,7 +13497,7 @@ static int glm_edge_engine_open(
model->c.vocab, model->c.hidden, io_bits);
model->lm_head = qt_load(model, "lm_head.weight",
model->c.vocab, model->c.hidden, io_bits);
model->final_norm = ld(model, "model.norm.weight");
model->final_norm = ld(model, "model.norm.weight", model->c.hidden);
char tokenizer_path[4096];
snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json",
options->model_dir);
Expand Down
43 changes: 25 additions & 18 deletions c/inkling.c
Original file line number Diff line number Diff line change
Expand Up @@ -718,11 +718,17 @@ static const char *embed_norm_name(shards *S) {
return NULL;
}

static float *load_t(Model *m, const char *name) {
/* `want` is the element count the forward pass indexes with (config dims); a
* short tensor used to be read past its end at inference (see qwen36 load_t_n). */
static float *load_t(Model *m, const char *name, int64_t want) {
int64_t n = st_numel(&m->S, name);
if (n < 0) { fprintf(stderr, "missing %s\n", name); exit(1); }
if (n != want) {
fprintf(stderr, "%s: %lld elements, config implies %lld -- refusing\n",
name, (long long)n, (long long)want); exit(1);
}
float *p = falloc(n);
st_read_f32(&m->S, name, p, 0);
st_read_f32_cap(&m->S, name, p, want, 0);
return p;
}
static float load_scalar(Model *m, const char *name, float dflt) {
Expand Down Expand Up @@ -951,8 +957,8 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
#endif
if (load_boundaries) {
m->embed = load_w(m, "model.embed_tokens.weight", 0);
{ const char *en = embed_norm_name(&m->S); m->embed_norm = en ? load_t(m, en) : NULL; }
m->final_norm = load_t(m, "model.norm.weight");
{ const char *en = embed_norm_name(&m->S); m->embed_norm = en ? load_t(m, en, D) : NULL; }
m->final_norm = load_t(m, "model.norm.weight", D);
m->lm_head = load_w(m, "lm_head.weight", 1);
}
/* Inkling's audio "tower" is one embedding table + one RMSNorm. The int4
Expand All @@ -976,7 +982,7 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
exit(1);
}
m->audio_enc = load_w(m, "model.audio.encoder.weight", 0);
m->audio_norm = load_t(m, "model.audio.final_norm.weight");
m->audio_norm = load_t(m, "model.audio.final_norm.weight", D);
fprintf(stderr, "[audio] DMel encoder loaded (%d bins x %d levels -> D=%d)\n",
c->mel_bins, c->mel_vocab, D);
}
Expand All @@ -986,25 +992,26 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
char nm[320];
for (int i = layer_begin; i < layer_end; i++) {
Layer *l = &m->L[i];
#define LD(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm)
#define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm,want)
int64_t hd = L_HD(c,i), kvd = (int64_t)L_KV(c,i) * hd;
#define LDW(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_w(m,nm,1)
LD(in_ln, "input_layernorm.weight");
LD(post_ln,"post_attention_layernorm.weight");
LD(in_ln, "input_layernorm.weight", D);
LD(post_ln,"post_attention_layernorm.weight", D);
LDW(q, "self_attn.q_proj.weight"); LDW(k, "self_attn.k_proj.weight");
LDW(v, "self_attn.v_proj.weight"); LDW(r, "self_attn.r_proj.weight");
LDW(o, "self_attn.o_proj.weight");
LD(qn,"self_attn.q_norm.weight"); LD(kn,"self_attn.k_norm.weight");
LD(relp, "self_attn.rel_logits_proj.proj");
LD(k_cw, "self_attn.k_sconv.conv1d.weight");
LD(v_cw, "self_attn.v_sconv.conv1d.weight");
LD(a_cw, "attn_sconv.conv1d.weight");
LD(m_cw, "mlp_sconv.conv1d.weight");
LD(qn,"self_attn.q_norm.weight", hd); LD(kn,"self_attn.k_norm.weight", hd);
LD(relp, "self_attn.rel_logits_proj.proj", (int64_t)c->d_rel * L_EXT(c,i));
LD(k_cw, "self_attn.k_sconv.conv1d.weight", kvd * K);
LD(v_cw, "self_attn.v_sconv.conv1d.weight", kvd * K);
LD(a_cw, "attn_sconv.conv1d.weight", (int64_t)D * K);
LD(m_cw, "mlp_sconv.conv1d.weight", (int64_t)D * K);
if (!c->sparse[i]) {
LDW(dg, "mlp.gate_proj.weight"); LDW(du, "mlp.up_proj.weight"); LDW(dd, "mlp.down_proj.weight");
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.global_scale",i); l->dgs = load_scalar(m,nm,1.f);
} else {
LD(router, "mlp.gate.weight");
LD(rbias, "mlp.gate.e_score_correction_bias");
LD(router, "mlp.gate.weight", (int64_t)(c->n_experts + c->n_shared) * D);
LD(rbias, "mlp.gate.e_score_correction_bias", c->n_experts);
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.global_scale",i); l->rgs = load_scalar(m,nm,1.f);
LDW(sh_g, "mlp.shared_experts.gate_proj");
LDW(sh_u, "mlp.shared_experts.up_proj");
Expand Down Expand Up @@ -3269,8 +3276,8 @@ static int inkling_edge_engine_open(
}
model->embed = load_w(model, "model.embed_tokens.weight", 0);
const char *embed_norm = embed_norm_name(&model->S);
model->embed_norm = embed_norm ? load_t(model, embed_norm) : NULL;
model->final_norm = load_t(model, "model.norm.weight");
model->embed_norm = embed_norm ? load_t(model, embed_norm, model->c.hidden) : NULL;
model->final_norm = load_t(model, "model.norm.weight", model->c.hidden);
model->lm_head = load_w(model, "lm_head.weight", 1);
char tokenizer_path[4096];
snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json",
Expand Down
39 changes: 24 additions & 15 deletions c/olmoe.c
Original file line number Diff line number Diff line change
Expand Up @@ -482,11 +482,17 @@ static void load_cfg(Cfg *c, const char *snap) {
free(buf); free(arena);
}

static float *load_t(Model *m, const char *name) {
/* `want` is the element count the forward pass indexes with (config dims); a
* short tensor used to be read past its end at inference (see qwen36 load_t_n). */
static float *load_t(Model *m, const char *name, int64_t want) {
int64_t n = st_numel(&m->S, name);
if (n < 0) { fprintf(stderr, "missing %s\n", name); exit(1); }
if (n != want) {
fprintf(stderr, "%s: %lld elements, config implies %lld -- refusing\n",
name, (long long)n, (long long)want); exit(1);
}
float *p = falloc(n);
st_read_f32(&m->S, name, p, 0); /* densa: niente DONTNEED, resta residente */
st_read_f32_cap(&m->S, name, p, want, 0); /* densa: niente DONTNEED, resta residente */
return p;
}

Expand All @@ -507,21 +513,22 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
}
double t0 = now_s();
if (load_boundaries) {
m->embed = load_t(m, "model.embed_tokens.weight");
m->lm_head = load_t(m, "lm_head.weight");
m->final_norm = load_t(m, "model.norm.weight");
m->embed = load_t(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden);
m->lm_head = load_t(m, "lm_head.weight", (int64_t)c->vocab * c->hidden);
m->final_norm = load_t(m, "model.norm.weight", c->hidden);
}
m->L = calloc(c->n_layers, sizeof(Layer));
char nm[256];
for (int i = layer_begin; i < layer_end; i++) {
Layer *l = &m->L[i];
#define LD(field, suffix) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm)
LD(in_ln, "input_layernorm.weight");
LD(post_ln,"post_attention_layernorm.weight");
LD(q, "self_attn.q_proj.weight"); LD(k, "self_attn.k_proj.weight");
LD(v, "self_attn.v_proj.weight"); LD(o, "self_attn.o_proj.weight");
LD(qn,"self_attn.q_norm.weight"); LD(kn,"self_attn.k_norm.weight");
LD(gate, "mlp.gate.weight");
#define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,i); l->field = load_t(m,nm,want)
int64_t D = c->hidden;
LD(in_ln, "input_layernorm.weight", D);
LD(post_ln,"post_attention_layernorm.weight", D);
LD(q, "self_attn.q_proj.weight", D*D); LD(k, "self_attn.k_proj.weight", D*D);
LD(v, "self_attn.v_proj.weight", D*D); LD(o, "self_attn.o_proj.weight", D*D);
LD(qn,"self_attn.q_norm.weight", D); LD(kn,"self_attn.k_norm.weight", D);
LD(gate, "mlp.gate.weight", (int64_t)c->n_experts * D);
#undef LD
}
/* cap <= 0 is "you decide", the sentinel the launcher sends when nobody
Expand Down Expand Up @@ -2416,9 +2423,11 @@ static int olmoe_edge_engine_open(
"out of memory opening OLMoE Edge");
load_cfg(&engine->model.c, options->model_dir);
st_init(&engine->model.S, options->model_dir);
engine->model.embed = load_t(&engine->model, "model.embed_tokens.weight");
engine->model.lm_head = load_t(&engine->model, "lm_head.weight");
engine->model.final_norm = load_t(&engine->model, "model.norm.weight");
int64_t vd = (int64_t)engine->model.c.vocab * engine->model.c.hidden;
engine->model.embed = load_t(&engine->model, "model.embed_tokens.weight", vd);
engine->model.lm_head = load_t(&engine->model, "lm_head.weight", vd);
engine->model.final_norm = load_t(&engine->model, "model.norm.weight",
engine->model.c.hidden);
char tokenizer_path[4096];
snprintf(tokenizer_path, sizeof(tokenizer_path), "%s/tokenizer.json",
options->model_dir);
Expand Down
48 changes: 48 additions & 0 deletions c/tests/st_fixture.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
/* Shared by the size-trust tests: write a one-shard safetensors container whose
* tensors declare exactly the dtype, shape and byte count the case needs, then
* run each case in a child and inspect how it ended. */
#ifndef ST_FIXTURE_H
#define ST_FIXTURE_H
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <stdint.h>
#include <sys/wait.h>
typedef struct { const char *name, *dtype; int64_t numel, nbytes; } FxT;
static int fx_write(const char *dir, const FxT *t, int n){
char hdr[8192], path[512]; int len = 0; int64_t off = 0;
len += snprintf(hdr + len, sizeof hdr - len, "{");
for (int i = 0; i < n; i++, off += t[i - 1].nbytes)
len += snprintf(hdr + len, sizeof hdr - len,
"%s\"%s\":{\"dtype\":\"%s\",\"shape\":[%lld],\"data_offsets\":[%lld,%lld]}",
i ? "," : "", t[i].name, t[i].dtype, (long long)t[i].numel,
(long long)off, (long long)(off + t[i].nbytes));
len += snprintf(hdr + len, sizeof hdr - len, "}");
while (len % 8) hdr[len++] = ' ';
snprintf(path, sizeof path, "%s/model.safetensors", dir);
FILE *f = fopen(path, "wb"); if (!f) return -1;
uint64_t hl = (uint64_t)len; fwrite(&hl, 8, 1, f); fwrite(hdr, 1, (size_t)len, f);
for (int64_t i = 0; i < off; i++) fputc(0x3c, f); /* 0x3c3c3c3c = 0.0115f */
return fclose(f);
}
/* Run `self --child <c> <dir>`; returns the exit code and fills `log` with stderr. */
static int fx_child(const char *self, int c, const char *dir, char *log, size_t cap){
char err[600], cmd[1600];
snprintf(err, sizeof err, "%s/stderr.txt", dir);
snprintf(cmd, sizeof cmd, "\"%s\" --child %d \"%s\" >/dev/null 2>\"%s\"", self, c, dir, err);
int st = system(cmd), got = st >= 0 && WIFEXITED(st) ? WEXITSTATUS(st) : -1;
log[0] = 0; FILE *f = fopen(err, "rb");
if (f) { size_t r = fread(log, 1, cap - 1, f); log[r] = 0; fclose(f); }
remove(err);
return got;
}
static void fx_cleanup(const char *dir){
char p[600]; snprintf(p, sizeof p, "%s/model.safetensors", dir); remove(p); rmdir(dir);
}
/* A refusal is exit 1 naming the tensor, with no sanitizer report: ASan also
* exits 1, so the exit code alone cannot tell a refusal from a heap overflow. */
static int fx_refused(int got, const char *log, const char *tensor){
return got == 1 && strstr(log, tensor) && strstr(log, "refusing") &&
!strstr(log, "Sanitizer");
}
#endif
Loading