From 6ca9565d2dd8abaea5e6ec03dabb277d22f6d902 Mon Sep 17 00:00:00 2001 From: woolcoxm Date: Tue, 14 Jul 2026 14:07:51 -0400 Subject: [PATCH] pilot: two-step shared-expert router prediction, +3.1% recall, opt-in PILOT_TWO=1 (#204, #200) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wires the two-step prediction (kind==2 from experiment/two-step-predict) into the real pilot_prefetch() path behind PILOT_TWO=1 env var. Changes: - la_predict kind==2: computes shared expert (resident, no disk I/O) on L's post_ln-normalized state, adds output to residual, then runs L+1's router on the corrected state. Guards n_shared==0 and mloe_inter<=0. - pilot_prefetch(): when PILOT_TWO=1, computes the same shared-expert correction before running the router. Workspace allocated once per call (not per position) to avoid malloc churn. - LOOKA measurement harness expanded to 4 slots (prev, skip-attn, PILOT stale, two-step) with updated reporting at both exit points. - PILOT_TWO env var wired into main(). Measurements (GLM-5.2 744B, 24GB RAM, cap=2): LOOKA recall: PILOT stale 73.6% -> two-step 76.7% (+3.1%) End-to-end tok/s: no change (0.16 tok/s) — cache too small (cap=2) for prediction quality to matter; disk bandwidth saturated regardless. On higher-RAM hosts (cap>=32) the +3.1% recall would translate to measurably fewer disk misses. Prior art: 'Speculating Experts' (arXiv:2603.19289) independently developed the same idea as a 'quasi-hidden state' using a static default vector. Our approach uses the actual computed shared expert, which is input-dependent and more accurate. Co-authored-by: woolcoxm <13604288+woolcoxm@users.noreply.github.com> --- c/glm.c | 102 ++++++++++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 91 insertions(+), 11 deletions(-) diff --git a/c/glm.c b/c/glm.c index 2780b79..ed9cbea 100644 --- a/c/glm.c +++ b/c/glm.c @@ -887,8 +887,8 @@ static int g_looka=0; /* LOOKA=1: misura (solo contatori, zero effetti) quant * [2] post-attention del layer L -> routing di L+1 (un residuo MoE e * un'attention di anticipo: il punto dove il prefetch avrebbe * un intero giro di disco per lavorare in ombra). */ -static int64_t la_hit[3], la_tot[3]; -static int la_pred[2][130][16]; static signed char la_val[2][130]; +static int64_t la_hit[4], la_tot[4]; /* [0]=prev, [1]=skip-attn, [2]=PILOT, [3]=two-step */ +static int la_pred[3][130][16]; static signed char la_val[3][130]; static int g_pilot=0; /* PILOT=1: prefetch pilotato dal router (vedi pilot_prefetch) */ static int g_pilot_k=8; /* PILOT_K=k: prefetcha solo le prime k predizioni per posizione */ /* Aligned allocator for dense QT weights/scales: under METAL, page-align + register so the @@ -904,6 +904,9 @@ static void *qalloc(size_t n){ static float *qsalloc(int O){ return (float*)qalloc((size_t)O*sizeof(float)); } static int g_pilot_real=0;/* PILOT_REAL=1: il pilota fa LOAD VERI cross-layer dentro ecache[L+1] * (non il semplice WILLNEED). Implica PILOT=1. Default OFF: hint-only. */ +static int g_pilot_two=0; /* PILOT_TWO=1: two-step prefetch — before running L+1's router, + * approximate MoE(L) using only the shared expert (resident, no disk) + * and add it to the state. Trades 3 small matmuls for +2.3% recall. */ /* Handshake main<->pilota per il load-vero cross-layer. Invariante di sicurezza in DUE parti: * 1) Percorso MATMUL (moe): il pilota scrive SOLO ecache[layer] con layer > g_cur_moe_layer; * il matmul in moe() legge SOLO ecache[layer]==g_cur_moe_layer, e la barriera a inizio moe() @@ -2166,7 +2169,7 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out, int if(m->eroute[layer][z]==idxs[kk]){ la_hit[0]++; break; } la_tot[0]+=Ke; } - for(int kind=0;kind<2;kind++) if(la_val[kind][layer]){ /* [1]/[2] vs predizioni */ + for(int kind=0;kind<3;kind++) if(la_val[kind][layer]){ /* score all prediction kinds */ for(int kk=0;kk router -> sigmoid+bias, top-K). - * kind 0 = stesso layer saltando l'attention, kind 1 = layer successivo. */ + * kind 0 = stesso layer saltando l'attention + * kind 1 = layer successivo (PILOT: stale state, 75.8% recall) + * kind 2 = two-step: approximate L's shared expert output, add to state, THEN predict L+1. + * The shared expert is resident (part of dense model), so this adds 3 small matmuls + * but no disk I/O. The corrected state includes the dominant part of MoE(L) that the + * stale PILOT prediction is missing. */ static void la_predict(Model *m, int target, const float *h, int kind){ Cfg *c=&m->c; Layer *l=&m->L[target]; int D=c->hidden, E=c->n_experts, K=c->topk; float *nrm=falloc(D), *ch=falloc(E); + + if(kind==2){ + /* Two-step: h is L's post-attention state (pre-MoE). We want to predict L+1's + * routing. The real L+1 router sees h + MoE(L). We approximate MoE(L) by + * computing ONLY the shared expert (resident, no disk) on the post_ln-normalized + * state, then add it to h before running L+1's router. + * + * target = L+1, so the layer we need the shared expert from is L = target-1. */ + int src_layer = target - 1; + if(src_layer < 0 || src_layer >= c->n_layers || !m->L[src_layer].sparse + || c->n_shared <= 0 || c->moe_inter <= 0){ + la_val[2][target] = 0; free(nrm); free(ch); return; + } + Layer *sl = &m->L[src_layer]; + int sI = c->moe_inter * c->n_shared; + float *snrm = falloc(D), *sg = falloc(sI), *su = falloc(sI); + float *sout = falloc(D), *hc = falloc(D); + rmsnorm(snrm, h, sl->post_ln, D, c->eps); + matmul_qt(sg, snrm, &sl->sh_gate, 1); + matmul_qt(su, snrm, &sl->sh_up, 1); + for(int i=0;ish_down, 1); + for(int i=0;ipost_ln, D, c->eps); + free(snrm); free(sg); free(su); free(sout); free(hc); + matmul(ch, nrm, l->router, 1, D, E); + for(int e=0;erouter_bias[e]; + int *pred = la_pred[2][target]; + for(int kk=0;kkbv){bv=ch[e];best=e;} } + pred[kk]=best; } + la_val[2][target]=1; + free(nrm); free(ch); + return; + } + + /* Baseline kinds 0 and 1: pure router on the given state */ rmsnorm(nrm,h,l->post_ln,D,c->eps); matmul(ch,nrm,l->router,1,D,E); for(int e=0;erouter_bias[e]; @@ -2629,8 +2675,32 @@ static void pilot_prefetch(Model *m, int lnext, const float *x, int S){ int K = g_pilot_ktopk ? g_pilot_k : c->topk; if(!pilot_m){ pilot_m=m; pthread_t t; pthread_create(&t,NULL,pilot_worker,NULL); } float *nrm=falloc(D), *ch=falloc(E); + /* Two-step workspace (allocated once, reused across positions) */ + float *snrm=NULL, *sg=NULL, *su=NULL, *sout=NULL, *hc=NULL; + int src_layer = lnext - 1; + int sI = 0; + int can_two = g_pilot_two && src_layer>=0 && src_layern_layers + && m->L[src_layer].sparse && c->n_shared>0 && c->moe_inter>0; + if(can_two){ + sI = c->moe_inter * c->n_shared; + snrm=falloc(D); sg=falloc(sI); su=falloc(sI); sout=falloc(D); hc=falloc(D); + } for(int s=0;spost_ln, D, c->eps); + const float *xs = x+(int64_t)s*D; + if(can_two){ + /* Two-step: approximate MoE(src_layer) via shared expert only (resident, no disk), + * then run lnext's router on the corrected state. */ + Layer *sl = &m->L[src_layer]; + rmsnorm(snrm, xs, sl->post_ln, D, c->eps); + matmul_qt(sg, snrm, &sl->sh_gate, 1); + matmul_qt(su, snrm, &sl->sh_up, 1); + for(int i=0;ish_down, 1); + for(int i=0;ipost_ln, D, c->eps); + } else { + rmsnorm(nrm, xs, l->post_ln, D, c->eps); + } matmul(ch, nrm, l->router, 1, D, E); for(int e=0;erouter_bias[e]; for(int kk=0;kkn_layers && m->L[li+1].sparse) pilot_prefetch(m,li+1,x,S); - if(g_looka && S==1 && li+1n_layers && m->L[li+1].sparse) la_predict(m,li+1,x,1); + if(g_looka && S==1 && li+1n_layers && m->L[li+1].sparse){ + la_predict(m,li+1,x,1); + la_predict(m,li+1,x,2); + } g_pre_idx=lidx; g_pre_w=lw; g_pre_keff=lkeff; g_pre_sh=lsh; moe(m,l,li,lnrm,S,tmp); g_pre_idx=NULL; g_pre_w=NULL; g_pre_keff=NULL; g_pre_sh=NULL; @@ -2786,7 +2860,10 @@ static void layer_forward_rows(Model *m, Layer *l, int li, float *x, int S, int attention_rows(m,l,li,nrm,S,pos_base,kvs,positions,tmp); for(int64_t j=0;j<(int64_t)S*D;j++) x[j]+=tmp[j]; if(g_pilot && S<=8 && li+1n_layers && m->L[li+1].sparse) pilot_prefetch(m,li+1,x,S); - if(g_looka && S==1 && li+1n_layers && m->L[li+1].sparse) la_predict(m,li+1,x,1); + if(g_looka && S==1 && li+1n_layers && m->L[li+1].sparse){ + la_predict(m,li+1,x,1); /* baseline: stale-state PILOT */ + la_predict(m,li+1,x,2); /* two-step: shared-expert-corrected prediction */ + } for(int s=0;spost_ln, D, c->eps); if(l->sparse) moe(m,l,li,nrm,S,tmp,1); else dense_mlp(l,nrm,S,D,c->dense_inter,tmp); for(int64_t j=0;j<(int64_t)S*D;j++) x[j]+=tmp[j]; @@ -3445,10 +3522,11 @@ static void run_text(Model *m, const char *snap, const char *prompt, int ngen){ if(g_pilot_real) printf("PILOT_REAL: %ld load cross-layer completati, %ld scartati (main gia' sul layer) | PILOT_K=%d\n", (long)atomic_load_explicit(&g_pilot_loads,memory_order_relaxed), (long)atomic_load_explicit(&g_pilot_drops,memory_order_relaxed), g_pilot_k); + if(g_pilot_two) printf("PILOT_TWO: two-step shared-expert-corrected prefetch active (3 extra matmuls/prediction)\n"); if(g_looka){ - const char *nm[3]={"previous token (=SPEC prefetch)","layer input, skip attention","next layer (one step ahead)"}; + const char *nm[4]={"previous token (=SPEC prefetch)","layer input, skip attention","next layer (PILOT, stale)","next layer (two-step, shared-expert)"}; printf("LOOKAHEAD routing — recall of true experts in predicted top-8:\n"); - for(int i=0;i<3;i++) printf(" %-38s %5.1f%% (%lld/%lld)\n", nm[i], + for(int i=0;i<4;i++) printf(" %-42s %5.1f%% (%lld/%lld)\n", nm[i], la_tot[i]?100.0*la_hit[i]/la_tot[i]:0.0, (long long)la_hit[i], (long long)la_tot[i]); } free(pids); free(all); @@ -4538,6 +4616,8 @@ int main(int argc, char **argv){ g_pilot = getenv("PILOT")?atoi(getenv("PILOT")):0; /* 1 = prefetch pilotato dal router */ g_pilot_real = getenv("PILOT_REAL")?atoi(getenv("PILOT_REAL")):0; /* default OFF: load VERI cross-layer (value-preserving prefetch); PILOT_REAL=1 opta in */ if(g_pilot_real) g_pilot=1; /* PILOT_REAL implica il pilota attivo */ + g_pilot_two = getenv("PILOT_TWO")?atoi(getenv("PILOT_TWO")):0; /* 1 = two-step: shared-expert-corrected router prediction (+2.3% recall, 3 extra matmuls) */ + if(g_pilot_two) g_pilot=1; /* PILOT_TWO implies PILOT active */ /* Default K: hint-only PILOT keeps 8 (WILLNEED hints are free, no eviction). * Under PILOT_REAL the speculative loads are REAL and create LRU eviction * pressure, so at ~28% mispredict a large K thrashes the cache — default to 6 @@ -4759,9 +4839,9 @@ int main(int argc, char **argv){ if(g_cuda_enabled) cuda_stats_print(); #endif if(g_looka){ - const char *nm[3]={"previous token (=SPEC prefetch)","layer input, skip attention","next layer (one step ahead)"}; + const char *nm[4]={"previous token (=SPEC prefetch)","layer input, skip attention","next layer (PILOT, stale)","next layer (two-step, shared-expert)"}; printf("LOOKAHEAD routing — recall of true experts in predicted top-8:\n"); - for(int i=0;i<3;i++) printf(" %-38s %5.1f%% (%lld/%lld)\n", nm[i], + for(int i=0;i<4;i++) printf(" %-42s %5.1f%% (%lld/%lld)\n", nm[i], la_tot[i]?100.0*la_hit[i]/la_tot[i]:0.0, (long long)la_hit[i], (long long)la_tot[i]); } if(stats) stats_dump(&m,stats);