From 78c675eb8b898e8f8c62ebcfe52e7b9034e03233 Mon Sep 17 00:00:00 2001 From: woolcoxm <13604288+woolcoxm@users.noreply.github.com> Date: Wed, 15 Jul 2026 18:33:25 -0400 Subject: [PATCH] cuda: fix pipe2 profile double-count + add COLI_CUDA_MTP opt-in (#292) Two issues found by the diagnostic sweep in #292: 1. Profile double-count in pipe_layer_sparse: the outer t_emm span wrapped both the shared-expert GPU dispatch AND the moe() call, but moe() self-times its own t_emm internally (6 accumulation sites). Routed-expert matmul was counted twice -> accounted > elapsed -> 'other' went negative, and expert-matmul showed an inflated ~120s. Fix: split into two narrow spans around only the GPU work moe() does NOT cover (shared-expert dispatch, routed upload+add), letting moe() self-time as all other callers already do. Verified: 'other' now positive (14-17s), expert-matmul realistic (5-6s). 2. MTP blanket-disabled under CUDA (g_draft=0 when g_cuda_enabled). This was a conservative guard from #163 before the root cause (cold-expert fused-pair + IDOT kernel divergence) was fully diagnosed. GPU-resident experts have no divergence; the cold subset still does but achieves 30-50% acceptance anyway (matching non-CUDA builds). Add COLI_CUDA_MTP=1 opt-in so users can test speculation under CUDA. Default unchanged. Verified: COLI_CUDA_MTP=1 -> MTP ACTIVE (draft=3), 44% acceptance, 3 forwards for 8 tokens (vs 7 without MTP). Refs #292 #163 --- c/glm.c | 25 +++++++++++++++++++++---- 1 file changed, 21 insertions(+), 4 deletions(-) diff --git a/c/glm.c b/c/glm.c index 5880f19..83f9a7c 100644 --- a/c/glm.c +++ b/c/glm.c @@ -3093,18 +3093,26 @@ static int pipe_layer_sparse(Model *m, Layer *l, int li, float *x_dev, int S, in * No pipe_sync at the end: the next layer's pipe_download (sync cudaMemcpy) * provides the implicit sync point. The fallback path (caller downloads x_dev) * also uses pipe_download which syncs. This lets GPU work chain across layers - * without a per-layer stall. */ + * without a per-layer stall. + * + * Profiling: moe() self-times its own t_emm (routed expert matmul). We time only + * the GPU work that moe() does NOT cover: the shared-expert dispatch and the + * routed-expert upload+add. Previously a single outer span wrapped everything + * including moe(), double-counting the routed-expert time and driving the + * profile's "other" bucket negative (#292). */ double te=now_s(); if(!coli_cuda_pipe_gemm(l->sh_gate.cuda,sg_d,nrm_d,S)) return 0; if(!coli_cuda_pipe_gemm(l->sh_up.cuda,su_d,nrm_d,S)) return 0; if(!coli_cuda_pipe_silu_mul(dev,sg_d,su_d,(size_t)S*sI)) return 0; if(!coli_cuda_pipe_gemm(l->sh_down.cuda,y_d,sg_d,S)) return 0; if(!coli_cuda_pipe_add(dev,x_dev,y_d,(size_t)S*D)) return 0; /* shared residual (async) */ + m->t_emm += now_s()-te; /* shared-expert GPU dispatch only */ /* expert routed su CPU/gruppi GPU come oggi (shared saltata: la fa il device) */ - moe(m,l,li,nrm_host,S,out_host,0); + moe(m,l,li,nrm_host,S,out_host,0); /* self-times its own t_emm */ + te=now_s(); if(!coli_cuda_pipe_upload(dev,y_d,out_host,xb)) return 0; /* sync: waits for moe */ if(!coli_cuda_pipe_add(dev,x_dev,y_d,(size_t)S*D)) return 0; /* routed residual (async) */ - m->t_emm+=now_s()-te; + m->t_emm += now_s()-te; /* routed-expert upload + add only */ return 1; } #endif @@ -5213,7 +5221,16 @@ int main(int argc, char **argv){ Model m; double t0=now_s(); model_init(&m,snap,cap,ebits,dbits); if(g_draft<0){ #ifdef COLI_CUDA - g_draft = (m.has_mtp&&!g_cuda_enabled) ? 3 : 0; + /* MTP is disabled under CUDA by default: cold (streaming) experts still + * run on the CPU, where the S==1 fused-pair kernel and the S>=2 IDOT + * kernel diverge in FP accumulation order, collapsing draft acceptance + * (#163). GPU-resident experts have no divergence, but the cold subset + * always exists on a single 16 GB card. COLI_CUDA_MTP=1 opts in for + * users who want to test speculation under CUDA — the #163 thread shows + * acceptance can still reach 30-50% even with the cold-expert mismatch. + * See #292 for the diagnostic sweep that identified this. */ + int cuda_mtp = getenv("COLI_CUDA_MTP") ? atoi(getenv("COLI_CUDA_MTP")) : 0; + g_draft = (m.has_mtp && (!g_cuda_enabled || cuda_mtp)) ? 3 : 0; #else g_draft = m.has_mtp ? 3 : 0; #endif