Merge remote-tracking branch 'upstream/dev' into feat/gpu-backend-hardening

This commit is contained in:
noobdev-ph
2026-07-18 02:55:12 +08:00
35 changed files with 2562 additions and 649 deletions
+132
View File
@@ -0,0 +1,132 @@
/* Microbenchmark: old (full-qsort) vs new (quickselect partial-select) DSA top-keep.
*
* This is NOT a unit test -- test_dsa_select.c proves correctness. This measures the
* headline claim of #356: that replacing the O(nk log nk) qsort over all nk context
* scores with an O(nk) partial_select_desc is materially faster per call, which is
* the win the issue was opened for -- and that the win GROWS with context length
* (because quickselect is linear average, qsort is n-log-n).
*
* It re-implements the OLD top-keep inline (qsort + threshold + scans) on a private
* buffer so the A/B runs in one process, same inputs, same warm caches -- a controlled
* comparison. It calls the REAL (new) partial_select_desc via the include-glm.c
* pattern, replicating the production threshold derivation + scans.
*
* Methodology (chosen to be honest, not to flatter the change):
* - keep = 2048 (the real GLM-5.2 index_topk), nk swept across context lengths from
* the 2049 activation boundary up to 65536 (a long conversation).
* - Three score shapes: (a) realistic peaked -- a few hot keys, long tail, the shape
* real DSA attention scores take; (b) uniform random -- no structure; (c) a plateau
* of ties, to exercise the boundary-membership path.
* - Each (shape, nk) is timed over N_REPEAT=2000 iterations, with the scores frozen
* so both algorithms do IDENTICAL work. We report median ns/call and the new/old
* ratio. A warmup pass primes caches before timing.
*
* Run: make tests/bench_dsa_select && ./tests/bench_dsa_select (not in TEST_BINS)
*/
#define main coli_glm_main_unused
#include "../glm.c"
#undef main
#include <math.h>
#include <stdint.h>
/* ---- the OLD algorithm, verbatim from dev before #356, on a private buffer ---- */
static int cmp_pdesc_old(const void *a, const void *b){
float x=*(const float*)a, y=*(const float*)b; return x<y?1:x>y?-1:0; }
static void keep_old(const float *isc, int nk, int keep, int *dst, int *nd_out){
float *tmp=malloc((size_t)nk*sizeof(float));
memcpy(tmp,isc,(size_t)nk*sizeof(float));
qsort(tmp,(size_t)nk,sizeof(float),cmp_pdesc_old);
float thr=tmp[keep-1]; int nd=0;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]>thr) dst[nd++]=t;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]==thr) dst[nd++]=t;
free(tmp); *nd_out=nd;
}
/* ---- timing: median of N_REPEAT runs in ns/call, sorted ascending ---- */
#define N_REPEAT 2000
static double bench_ns(void (*fn)(const float*,int,int,int*,int*),
const float *isc, int nk, int keep){
static double ts[N_REPEAT]; int *dst=malloc((size_t)nk*sizeof(int)); int nd;
for(int r=0;r<N_REPEAT;r++){
double t0=now_s();
fn(isc,nk,keep,dst,&nd);
ts[r]=(now_s()-t0)*1e9;
}
for(int a=1;a<N_REPEAT;a++){ double k=ts[a]; int b=a-1;
while(b>=0 && ts[b]>k){ ts[b+1]=ts[b]; b--; } ts[b+1]=k; }
free(dst);
return ts[N_REPEAT/2];
}
/* the NEW algorithm calls the real partial_select_desc + replicates the production
* threshold derivation and position scans. */
static void keep_new(const float *isc, int nk, int keep, int *dst, int *nd_out){
float *tmp=malloc((size_t)nk*sizeof(float));
memcpy(tmp,isc,(size_t)nk*sizeof(float));
partial_select_desc(tmp,nk,keep);
float thr=tmp[0]; for(int t=1;t<keep;t++) if(tmp[t]<thr) thr=tmp[t];
int nd=0;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]>thr) dst[nd++]=t;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]==thr) dst[nd++]=t;
free(tmp); *nd_out=nd;
}
/* deterministic score fill for three shapes */
static uint32_t brng = 0xA5A5A5A5u;
static double brand(void){ brng ^= brng << 13; brng ^= brng >> 17; brng ^= brng << 5;
return (double)(brng >> 8) * (1.0 / 16777216.0); }
static void fill_realistic(float *isc, int nk){ /* few hot, long distinct tail */
for(int i=0;i<nk;i++) isc[i]=(float)(-1.0 - brand()*4.0);
isc[0]=3.f; isc[nk/50<nk?nk/50:nk-1]=1.f; isc[nk/200<nk?nk/200:nk-1]=0.5f;
}
static void fill_uniform(float *isc, int nk){ /* no structure */
for(int i=0;i<nk;i++) isc[i]=(float)(brand()*1000.0);
}
static void fill_plateau(float *isc, int nk){ /* tie blocks -> boundary path */
for(int i=0;i<nk;i++) isc[i]=(float)(-(double)(i/7));
}
int main(void){
int keep = 2048; /* GLM-5.2 index_topk */
int nks[] = {2049, 4096, 8192, 16384, 32768, 65536};
float *isc = malloc((size_t)65536*sizeof(float));
int *dst = malloc((size_t)65536*sizeof(int)); int nd;
struct { const char *name; void (*fill)(float*,int); } shapes[] = {
{ "realistic", fill_realistic },
{ "uniform", fill_uniform },
{ "plateau", fill_plateau },
};
printf("bench_dsa_select: DSA top-keep, old (qsort) vs new (partial-select) keep=%d\n", keep);
printf("%-12s %7s %14s %14s %9s\n", "shape", "nk", "old ns/call", "new ns/call", "speedup");
printf("------------------------------------------------------------------------\n");
for(size_t sh=0; sh<sizeof(shapes)/sizeof(shapes[0]); sh++){
for(size_t ni=0; ni<sizeof(nks)/sizeof(nks[0]); ni++){
int nk=nks[ni];
shapes[sh].fill(isc,nk);
/* warmup both paths so caches/branch predictors are primed */
for(int w=0; w<50; w++){ keep_old(isc,nk,keep,dst,&nd); keep_new(isc,nk,keep,dst,&nd); }
/* sanity: both must keep exactly `keep` (correctness is test_dsa_select's
* job, but a count divergence here would make the timing meaningless) */
keep_old(isc,nk,keep,dst,&nd); int na=nd;
keep_new(isc,nk,keep,dst,&nd); int nb=nd;
if(na!=keep || nb!=keep){
printf("%-12s %7d (BAD COUNTS: old=%d new=%d, skipped)\n",
shapes[sh].name, nk, na, nb);
continue;
}
double t_old=bench_ns(keep_old,isc,nk,keep);
double t_new=bench_ns(keep_new,isc,nk,keep);
printf("%-12s %7d %14.0f %14.0f %8.2fx\n",
shapes[sh].name, nk, t_old, t_new, t_old/t_new);
}
printf("\n");
}
printf("bench_dsa_select: done (lower ns is better; speedup = old/new)\n");
free(isc); free(dst);
return 0;
}
+131
View File
@@ -0,0 +1,131 @@
/* Microbenchmark: old (full-vocab qsort) vs new (heap partial-select) top-p truncation.
*
* This is NOT a unit test -- test_topp.c proves correctness. This measures the headline
* claim of #335: that replacing the O(V log V) qsort over all 151936 vocab entries with
* an O(V) heapify + k*log-V pops is materially faster per call, which is the win the issue
* was opened for.
*
* It re-implements the OLD dist_build inline (qsort + scan) on a private buffer so the A/B
* runs in one process, same inputs, same warm caches -- a controlled comparison. It calls
* the REAL (new) dist_build via the include-glm.c pattern on the global g_pbuf.
*
* Methodology (chosen to be honest, not to flatter the change):
* - V = 151936 (the actual GLM-5.2 vocab), g_nuc swept across the values that matter
* for serving: 0.5 / 0.9 (serve default) / 0.95 / 0.99.
* - Three logit shapes: (a) realistic peaked -- one hot token, long exponential tail,
* the shape real language-model logits take; (b) uniform -- worst case for the heap,
* maximum pop count; (c) a plateau of ties, to exercise the tie path.
* - Each (shape, nuc) is timed over N_REPEAT=2000 iterations, with the RNG/logits frozen
* so both algorithms do IDENTICAL work. We report median ns/call and the new/old ratio.
* - A warmup pass primes caches before timing.
*
* Run: make tests/bench_topp && ./tests/bench_topp (not in TEST_BINS -- not a gate)
*/
#define main coli_glm_main_unused
#include "../glm.c"
#undef main
#include <math.h>
#include <stdint.h>
/* ---- the OLD algorithm, verbatim from dev before #354, on a private buffer ---- */
static float *s_pbuf; static int *s_pidx; static double *s_ref;
static int cmp_pdesc_old(const void *a, const void *b){
double pa = s_ref[*(const int*)a], pb = s_ref[*(const int*)b];
return pa < pb ? 1 : pa > pb ? -1 : 0; }
static void dist_build_old(const float *lo, int V, double temp, double nuc){
double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i];
double s = 0, invt = 1.0 / (temp > 1e-4 ? temp : 1e-4);
for (int i = 0; i < V; i++){ s_ref[i] = exp((lo[i]-mx)*invt); s += s_ref[i]; }
for (int i = 0; i < V; i++) s_ref[i] /= s;
if (nuc > 0 && nuc < 1.0){
for (int i = 0; i < V; i++) s_pidx[i] = i;
qsort(s_pidx, V, sizeof(int), cmp_pdesc_old);
double cum = 0; int keep = V;
for (int i = 0; i < V; i++){ cum += s_ref[s_pidx[i]]; if (cum >= nuc){ keep = i+1; break; } }
double s2 = 0;
for (int i = keep; i < V; i++) s_ref[s_pidx[i]] = 0;
for (int i = 0; i < keep; i++) s2 += s_ref[s_pidx[i]];
for (int i = 0; i < keep; i++) s_ref[s_pidx[i]] /= s2;
}
(void)s_pbuf;
}
/* ---- timing: median of N_REPEAT runs in ns/call, sorted ascending ---- */
#define N_REPEAT 2000
static double bench_ns(void (*fn)(const float*,int,double,double),
const float *lo, int V, double temp, double nuc){
static double ts[N_REPEAT];
for (int r = 0; r < N_REPEAT; r++){
double t0 = now_s();
fn(lo, V, temp, nuc);
ts[r] = (now_s() - t0) * 1e9;
}
/* insertion sort the N_REPEAT samples (small), take median */
for (int a = 1; a < N_REPEAT; a++){ double k = ts[a]; int b = a-1;
while (b >= 0 && ts[b] > k){ ts[b+1] = ts[b]; b--; } ts[b+1] = k; }
return ts[N_REPEAT/2];
}
/* the NEW algorithm is the real dist_build, but it writes g_pbuf (not a private buf).
* Wrap it so the bench signature matches, and set the globals it reads. */
static void dist_build_new(const float *lo, int V, double temp, double nuc){
g_temp = (float)temp; g_nuc = (float)nuc;
dist_build(lo, V);
}
/* deterministic logit fill for three shapes */
static uint32_t brng = 0xA5A5A5A5u;
static double brand(void){ brng ^= brng << 13; brng ^= brng >> 17; brng ^= brng << 5;
return (double)(brng >> 8) * (1.0 / 16777216.0); }
static void fill_realistic(float *lo, int V){ /* one hot, exponential tail -- like real logits */
for (int i = 0; i < V; i++) lo[i] = (float)(-4.0 * brand() - (double)i * 0.0001);
lo[0] = 6.f; lo[V/50] = 4.f; lo[V/200] = 3.f;
}
static void fill_uniform(float *lo, int V){ /* worst case for the heap: max pop count */
for (int i = 0; i < V; i++) lo[i] = 0.f;
}
static void fill_plateau(float *lo, int V){ /* ties: blocks of equal value */
for (int i = 0; i < V; i++) lo[i] = (float)(-(double)(i / 50));
}
int main(void){
int V = 151936;
float *lo = malloc((size_t)V * sizeof(float));
s_ref = malloc((size_t)V * sizeof(double));
s_pidx = malloc((size_t)V * sizeof(int));
/* force the new dist_build to allocate g_pbuf/g_pidx at full V once */
g_temp = 0.7f; g_nuc = 0.9f; dist_build(lo, V);
double temp = 0.7;
struct { const char *name; void (*fill)(float*,int); } shapes[] = {
{ "realistic", fill_realistic },
{ "uniform", fill_uniform },
{ "plateau", fill_plateau },
};
double nucs[] = { 0.5, 0.9, 0.95, 0.99 };
printf("bench_topp: top-p truncation, old (qsort) vs new (heap) V=%d temp=%.2f\n", V, temp);
printf("%-12s %6s %14s %14s %9s %9s\n", "shape", "nuc", "old ns/call", "new ns/call", "speedup", "keep");
printf("-----------------------------------------------------------------------------\n");
for (size_t sh = 0; sh < sizeof(shapes)/sizeof(shapes[0]); sh++){
shapes[sh].fill(lo, V);
for (size_t ni = 0; ni < sizeof(nucs)/sizeof(nucs[0]); ni++){
double nuc = nucs[ni];
/* warmup both paths so caches/branch predictors are primed */
for (int w = 0; w < 50; w++){ dist_build_old(lo, V, temp, nuc); dist_build_new(lo, V, temp, nuc); }
double t_old = bench_ns(dist_build_old, lo, V, temp, nuc);
double t_new = bench_ns(dist_build_new, lo, V, temp, nuc);
/* keep count = non-zero entries the new path leaves (== old's keep) */
int keep = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) keep++;
printf("%-12s %6.2f %14.0f %14.0f %8.2fx %9d\n",
shapes[sh].name, nuc, t_old, t_new, t_old / t_new, keep);
}
printf("\n");
}
printf("bench_topp: done (lower ns is better; speedup = old/new)\n");
free(lo); free(s_ref); free(s_pidx);
return 0;
}
+223
View File
@@ -0,0 +1,223 @@
/* DSA top-keep partial-select: the quickselect rewrite (#356) must produce a
* BIT-IDENTICAL kept-position set to the old full-vocab qsort, for every score
* shape the attention indexer can see.
*
* Why this test exists (#356): attention_rows() selects the top-`keep` context
* keys (index_topk=2048 on GLM-5.2) to attend to. It previously did this by
* full-qsorting all `nk` scores (O(nk log nk)) and reading tmp[keep-1] as the
* threshold. It now does a partial_select_desc (quickselect, O(nk) average) and
* takes the threshold as the min of the selected top-keep block. The contract is
* subtle but STRONGER than test_topp's:
*
* The two position-order scans that build dst[] --
* for t: if isc[t] > thr -> keep (strictly above threshold)
* for t: if isc[t] == thr -> keep (ties, in position order)
* -- are UNCHANGED by the rewrite. So if the threshold value is identical,
* the kept-position set is identical element-by-element (not just as a
* multiset, which is all the unstable sampling heap in #335 could promise).
*
* Strategy: drive the REAL partial_select_desc (via the include-glm.c pattern)
* and replicate the production threshold derivation + scans, then compare the
* resulting dst[] against an INDEPENDENT reference that re-implements the OLD
* algorithm (full qsort + tmp[keep-1] threshold) on a private buffer. The kept
* sets must be element-wise equal on every shape, including tie plateaus where
* the boundary membership is decided by the position scan.
*
* We also directly unit-test partial_select_desc's partition invariant: after
* the call, max(a[keep..n)) <= min(a[0..keep)) -- i.e. the keep largest really
* did land in the prefix. This catches a broken quickselect even before the
* end-to-end comparison.
*
* In-memory only (no scratch files), so it builds clean on the Windows MinGW CI
* job without the unmerged compat shim. */
#define main coli_glm_main_unused
#include "../glm.c"
#undef main
#include <math.h>
static int g_nfails = 0;
#define FAIL(fmt, ...) do { \
fprintf(stderr, " FAIL [%s nk=%d keep=%d shape=%s]: " fmt "\n", \
label, nk, keep, shape_name, ##__VA_ARGS__); \
g_nfails++; \
return; \
} while (0)
/* ---- independent reference: the OLD algorithm (full qsort + tmp[keep-1]) ---- */
/* qsort comparator matching the production cmp_fdesc exactly (desc, unstable). */
static int cmp_ref_desc(const void *a, const void *b){
float x=*(const float*)a, y=*(const float*)b; return x<y?1:x>y?-1:0; }
/* Reproduce the OLD glm.c:2589-2596 exactly: copy, qsort desc, threshold =
* tmp[keep-1], then the two position-order scans into dst[]. Returns nd. */
static int keep_old(const float *isc, int nk, int keep, int *dst){
float *tmp=malloc((size_t)nk*sizeof(float));
memcpy(tmp,isc,(size_t)nk*sizeof(float));
qsort(tmp,(size_t)nk,sizeof(float),cmp_ref_desc);
float thr=tmp[keep-1];
int nd=0;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]>thr) dst[nd++]=t;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]==thr) dst[nd++]=t;
free(tmp);
return nd;
}
/* Reproduce the NEW glm.c path: partial_select desc, threshold = min of the
* selected block, same two scans. Uses the REAL partial_select_desc from glm.c. */
static int keep_new(const float *isc, int nk, int keep, int *dst){
float *tmp=malloc((size_t)nk*sizeof(float));
memcpy(tmp,isc,(size_t)nk*sizeof(float));
partial_select_desc(tmp,nk,keep);
float thr=tmp[0]; for(int t=1;t<keep;t++) if(tmp[t]<thr) thr=tmp[t];
int nd=0;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]>thr) dst[nd++]=t;
for(int t=0;t<nk && nd<keep;t++) if(isc[t]==thr) dst[nd++]=t;
free(tmp);
return nd;
}
/* ---- direct unit test of the partition invariant ---- */
static void check_partition(const char *label, const float *isc, int nk, int keep,
const char *shape_name){
float *tmp=malloc((size_t)nk*sizeof(float));
memcpy(tmp,isc,(size_t)nk*sizeof(float));
partial_select_desc(tmp,nk,keep);
/* invariant: every element of tmp[0..keep) is >= every element of tmp[keep..n).
* (>=, not >: equal values may sit on either side of the partition boundary,
* which is fine -- the threshold is the MIN of the prefix, and the position
* scan handles ties.) */
float top_min=INFINITY, tail_max=-INFINITY;
for(int i=0;i<keep;i++) if(tmp[i]<top_min) top_min=tmp[i];
for(int i=keep;i<nk;i++) if(tmp[i]>tail_max) tail_max=tmp[i];
if(!(top_min >= tail_max))
FAIL("partition invariant violated: top_min=%.9g < tail_max=%.9g", top_min, tail_max);
free(tmp);
}
/* ---- end-to-end: old vs new kept-set must be element-wise identical ---- */
static void check_case(const char *label, int nk, int keep, const char *shape_name,
const float *isc){
int *da=malloc((size_t)nk*sizeof(int));
int *db=malloc((size_t)nk*sizeof(int));
int na=keep_old(isc,nk,keep,da);
int nb=keep_new(isc,nk,keep,db);
/* 1. both keep exactly `keep` positions (the contract: keep the top-keep by
* count). A count mismatch is a real bug, not a tie artifact. */
if(na!=keep) FAIL("old kept %d, expected %d (old path is the reference)", na, keep);
if(nb!=keep) FAIL("new kept %d, expected %d", nb, keep);
if(na!=nb) FAIL("keep-count mismatch: old=%d new=%d", na, nb);
/* 2. element-wise identical dst[]. This is the strong contract: because the
* threshold is derived identically and the position-order scans are byte-
* for-byte the same, the kept SET and its ORDER must match exactly. (This
* is what makes #356 cleaner than #335, which was multiset-only.) */
int first_diff=-1;
for(int i=0;i<na;i++){ if(da[i]!=db[i]){ first_diff=i; break; } }
if(first_diff>=0)
FAIL("kept-set differs at index %d: old dst[%d]=%d new dst[%d]=%d",
first_diff, first_diff, da[first_diff], first_diff, db[first_diff]);
/* 3. also check the partition invariant directly (catches a subtly broken
* quickselect even if the threshold happened to come out right). */
check_partition(label,isc,nk,keep,shape_name);
free(da); free(db);
printf(" ok [nk=%d keep=%d shape=%s]\n", nk, keep, shape_name);
}
#undef FAIL
/* deterministic xorshift32 RNG (matches the test_i4_grouped.c / test_topp.c convention) */
static uint32_t rng_state = 0x12345678u;
static uint32_t xr(void){ rng_state ^= rng_state << 13; rng_state ^= rng_state >> 17;
rng_state ^= rng_state << 5; return rng_state; }
static double frand(void){ return (xr() >> 8) * (1.0 / 16777216.0); } /* [0,1) */
/* fill scores for a given shape. Shapes stress the threshold boundary and the
* quickselect's median-of-three pivot differently. */
static void fill_shape(float *isc, int nk, int shape){
switch(shape){
case 0: /* uniform random distinct (no ties): the clean contract case */
for(int i=0;i<nk;i++) isc[i]=(float)(frand()*1000.0); break;
case 1: /* peaked: a few hot, long distinct tail (realistic attention shape) */
for(int i=0;i<nk;i++) isc[i]=(float)(-1.0 - frand()*4.0);
isc[0]=3.f; if(nk>3) isc[nk/3]=1.f; if(nk>2) isc[nk/2]=0.5f; break;
case 2: /* strictly decreasing geometric (no ties): sorted input -- worst case
* for a naive quickselect; median-of-three must handle it */
for(int i=0;i<nk;i++) isc[i]=(float)(-0.001*(double)i); break;
case 3: /* strictly increasing (reverse-sorted): the other quickselect worst case */
for(int i=0;i<nk;i++) isc[i]=(float)(0.001*(double)i); break;
case 4: /* plateau ties: blocks of equal value -> boundary membership decided
* entirely by the position scan (exercises the ==thr path) */
for(int i=0;i<nk;i++) isc[i]=(float)(-(double)(i/7)); break;
case 5: /* all-equal: every value identical -> degenerate threshold, all kept
* via the ==thr scan; quickselect must not infinite-loop or corrupt */
for(int i=0;i<nk;i++) isc[i]=5.f; break;
}
}
int main(void){
/* Sizes around the real index_topk=2048 boundary, plus small cases that
* exercise the k>=n / k==1 / k==n edges. */
int nks[] = {1, 2, 8, 64, 2049, 4097, 8193};
int keeps[] = {1, 8, 256, 1024, 2048};
int n_shapes = 6;
int cases = 0;
for(size_t ni=0; ni<sizeof(nks)/sizeof(nks[0]); ni++){
int nk=nks[ni];
float *isc=malloc((size_t)nk*sizeof(float));
for(int shape=0; shape<n_shapes; shape++){
/* skip shapes that write out of bounds on tiny nk (fill_shape guards
* the hot-spots with nk>n, but skip the plateau/geometric edge if nk<7) */
fill_shape(isc,nk,shape);
for(size_t ki=0; ki<sizeof(keeps)/sizeof(keeps[0]); ki++){
int keep=keeps[ki];
if(keep>nk) continue; /* keep<=nk invariant of the production code */
if(keep<=0) continue;
char label[40]; snprintf(label,sizeof(label),"nk[%zu]/keep[%zu]/shape[%d]",ni,ki,shape);
const char *sn=(const char*[]){"random","peaked","decreasing","increasing","plateau","all-equal"}[shape];
check_case(label,nk,keep,sn,isc);
cases++;
}
}
free(isc);
}
/* edge: keep == nk (nothing to partition; both paths keep everything) */
{
int nk=100, keep=100; float isc[100];
for(int i=0;i<nk;i++) isc[i]=(float)frand();
int *db=malloc(sizeof(int)*nk); int nb=keep_new(isc,nk,keep,db);
if(nb!=nk){ fprintf(stderr," FAIL [keep==nk]: kept %d expected %d\n",nb,nk); g_nfails++; }
else printf(" ok [keep==nk nk=%d]\n",nk);
free(db); cases++;
}
/* edge: keep == 1 (threshold = the single max; quickselect must find it) */
{
int nk=500, keep=1; float isc[500];
for(int i=0;i<nk;i++) isc[i]=(float)frand();
int da[1],db[1]; int na=keep_old(isc,nk,keep,da), nb=keep_new(isc,nk,keep,db);
if(na!=1||nb!=1||da[0]!=db[0]){
fprintf(stderr," FAIL [keep==1]: old={%d (n=%d)} new={%d (n=%d)}\n",da[0],na,db[0],nb); g_nfails++; }
else printf(" ok [keep==1 argmax=%d]\n",db[0]); cases++;
}
/* edge: all-equal scores, keep in the middle -> every kept slot is a tie;
* the position scan must pick positions 0..keep-1 deterministically */
{
int nk=1000, keep=500; float isc[1000];
for(int i=0;i<nk;i++) isc[i]=3.14f;
int *db=malloc(sizeof(int)*nk); int nb=keep_new(isc,nk,keep,db);
int bad=0; for(int i=0;i<nb;i++) if(db[i]!=i) bad=1;
if(nb!=keep||bad){ fprintf(stderr," FAIL [all-equal keep=%d]: nb=%d bad=%d\n",keep,nb,bad); g_nfails++; }
else printf(" ok [all-equal keep=%d -> positions 0..%d]\n",keep,keep-1); cases++;
free(db);
}
printf("\ntest_dsa_select: %d cases run, %d failure(s)\n", cases, g_nfails);
if(g_nfails){ printf("test_dsa_select: FAIL\n"); return 1; }
printf("test_dsa_select: ok\n");
return 0;
}
+89
View File
@@ -0,0 +1,89 @@
/* st_pread_full: chunk loop + honest truncation errors.
* Built with -DST_PREAD_CHUNK=7 so a ~100-byte tensor takes many pread calls —
* exercising the loop that production only needs past 2^31 bytes (one pread
* caps there on Linux; big bf16 tensors exceed it). Also forks a child against
* a truncated shard and requires exit(1) with a "short read" message instead
* of the old perror("... : Success"). */
#define _GNU_SOURCE
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#ifndef _WIN32
#include <sys/wait.h>
#include <unistd.h>
#endif
#include "../st.h"
#define CHECK(condition) do { \
if (!(condition)) { \
fprintf(stderr, "%s:%d: check failed: %s\n", __FILE__, __LINE__, #condition); \
return 1; \
} \
} while (0)
static void write_snap(const char *dir, int truncate_bytes) {
char path[512];
snprintf(path, sizeof(path), "%s/model.safetensors", dir);
unsigned char data[96];
for (int i = 0; i < 96; i++) data[i] = (unsigned char)(i * 7 + 3);
const char *hdr = "{\"t\":{\"dtype\":\"U8\",\"shape\":[96],\"data_offsets\":[0,96]}}";
uint64_t hlen = strlen(hdr);
FILE *f = fopen(path, "wb");
fwrite(&hlen, 8, 1, f);
fwrite(hdr, 1, hlen, f);
fwrite(data, 1, (size_t)(96 - truncate_bytes), f);
fclose(f);
}
int main(void) {
/* relative to the CWD, per test_stops: MinGW .exe files resolve Windows
* paths and "/tmp" is not one */
char dir[] = "test_st_pread_XXXXXX";
if (!mkdtemp(dir)) { perror("mkdtemp"); return 1; }
/* 1) chunk loop: 96-byte tensor read 7 bytes at a time, content exact */
write_snap(dir, 0);
shards S; st_init(&S, dir);
unsigned char out[96] = {0};
st_read_raw(&S, "t", out, 0);
for (int i = 0; i < 96; i++) CHECK(out[i] == (unsigned char)(i * 7 + 3));
#ifndef _WIN32
/* 2) shard truncated AFTER st_init (init validates static bounds, so the
* pread path only fires when the file shrinks underneath a live handle):
* child must exit(1) with an honest message, not perror's "Success" */
char shard[512]; snprintf(shard, sizeof(shard), "%s/model.safetensors", dir);
struct stat sb; CHECK(stat(shard, &sb) == 0);
CHECK(truncate(shard, sb.st_size - 40) == 0);
int pipefd[2]; CHECK(pipe(pipefd) == 0);
pid_t pid = fork(); CHECK(pid >= 0);
if (pid == 0) {
dup2(pipefd[1], 2); close(pipefd[0]); close(pipefd[1]);
unsigned char buf[96];
st_read_raw(&S, "t", buf, 0); /* inherited handles; must exit(1) inside */
_exit(42); /* reaching here = bug */
}
close(pipefd[1]);
char err[512] = {0};
ssize_t n = read(pipefd[0], err, sizeof(err)-1); (void)n;
close(pipefd[0]);
int status = 0; waitpid(pid, &status, 0);
CHECK(WIFEXITED(status) && WEXITSTATUS(status) == 1);
CHECK(strstr(err, "short read") != NULL);
CHECK(strstr(err, "Success") == NULL);
#else
/* fork/pipe/truncate are POSIX; Windows still runs the chunk-loop check */
printf("test_st_pread: truncation subtest skipped on Windows\n");
#endif
char cmd[600];
#ifdef _WIN32
snprintf(cmd, sizeof(cmd), "rmdir /s /q %s", dir);
#else
snprintf(cmd, sizeof(cmd), "rm -rf %s", dir);
#endif
if (system(cmd)) {}
printf("test_st_pread: chunk loop + honest truncation error: ok\n");
return 0;
}
+280
View File
@@ -0,0 +1,280 @@
/* Top-p (nucleus) truncation in dist_build: the partial-select rewrite (#335) must be
* indistinguishable from the old full-vocab qsort for every shape dist_sample can see.
*
* Why this test exists (#335): dist_build() previously qsort-ed the entire 151936-entry
* vocab on every sampled token to find the few-hundred-token head whose cumulative mass
* reaches g_nuc. It now heapifies (O(V)) and pops only the head (k * O(log V)). The win
* is structural; the risk is a silent sampling-distribution change, because the contract
* is subtle:
*
* dist_sample() iterates g_pbuf[0..V-1] BY TOKEN ID and sums probabilities directly.
* So dist_build MUST leave g_pbuf indexed by id (never reordered) AND must zero every
* truncated tail entry -- merely excluding the tail from the head would leave mass on
* it and the sampled distribution would drift with no crash and no error.
*
* Strategy: drive the REAL dist_build (via the test_stops.c include-glm.c pattern) on a
* sweep of distributions and g_nuc values, and compare against an INDEPENDENT reference
* that re-implements the OLD algorithm (full qsort + zero-tail + renorm) in double on a
* private buffer. On shapes with no ties the renormalized head must be BIT-IDENTICAL to
* the reference (the issue's stated invariant: s2 accumulates in the same descending
* order). On tie shapes, where the unstable qsort already left ordering unspecified, we
* check multiset equality instead. Every shape also checks: exact-zero tails, head sums
* to 1.0, and a sane keep-count.
*
* No scratch files: the test runs entirely in memory (no mkdtemp), so it builds clean on
* the Windows MinGW CI job without the unmerged compat shim (#352). */
#define main coli_glm_main_unused
#include "../glm.c"
#undef main
#include <math.h>
static int g_nfails = 0;
/* pointer set by ref_build so cmp_ref_desc can read the current reference buffer
* (the qsort comparator gets no user-data argument in C). */
static const double *g_ref_p = NULL;
#define FAIL(fmt, ...) do { \
fprintf(stderr, " FAIL [%s V=%d nuc=%.3f shape=%s]: " fmt "\n", \
label, V, nuc, shape_name, ##__VA_ARGS__); \
g_nfails++; \
return; \
} while (0)
/* ---- independent reference: the OLD algorithm, in double, on a private buffer ------- */
/* Stable qsort by descending probability (ties broken by ascending index, which makes
* the reference deterministic regardless of the production comparator). */
static int cmp_ref_desc(const void *a, const void *b){
double pa = ((const double *)g_ref_p)[*(const int*)a];
double pb = ((const double *)g_ref_p)[*(const int*)b];
if (pa < pb) return 1;
if (pa > pb) return -1;
/* tie -> lower index first (stable, unlike the production comparator) */
return *(const int*)a - *(const int*)b;
}
/* Build the reference distribution into out[0..V-1] (indexed by token id), mirroring the
* old dist_build: softmax(lo/temp) truncated to top-p nuc, tail zeroed, head renormalized.
* Returns the keep-count through *keep_out. */
static void ref_build(const float *lo, int V, double temp, double nuc,
double *out, int *pidx, int *keep_out){
double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i];
double s = 0, invt = 1.0 / (temp > 1e-4 ? temp : 1e-4);
for (int i = 0; i < V; i++){ out[i] = exp((lo[i]-mx)*invt); s += out[i]; }
for (int i = 0; i < V; i++) out[i] /= s;
if (nuc > 0 && nuc < 1.0){
for (int i = 0; i < V; i++) pidx[i] = i;
qsort(pidx, V, sizeof(int), cmp_ref_desc);
double cum = 0; int keep = V;
for (int i = 0; i < V; i++){ cum += out[pidx[i]]; if (cum >= nuc){ keep = i+1; break; } }
double s2 = 0;
for (int i = keep; i < V; i++) out[pidx[i]] = 0;
for (int i = 0; i < keep; i++) s2 += out[pidx[i]];
for (int i = 0; i < keep; i++) out[pidx[i]] /= s2;
*keep_out = keep;
} else {
*keep_out = V;
}
}
/* count how many production g_pbuf entries are non-zero == the head size */
static int head_count(int V){
int n = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) n++; return n;
}
/* Run one case: load logits into g_pbuf via the real dist_build, compare to reference.
* shape_name is for diagnostics only. */
static void check_case(const char *label, int V, double nuc, const char *shape_name,
const float *lo){
/* reference on a private buffer */
double *ref = malloc((size_t)V * sizeof(double));
int *ridx = malloc((size_t)V * sizeof(int));
int ref_keep = 0;
g_ref_p = ref; /* cmp_ref_desc reads this */
ref_build(lo, V, g_temp, nuc, ref, ridx, &ref_keep);
/* production: drive the real dist_build (writes the global g_pbuf) */
g_nuc = (float)nuc;
dist_build(lo, V);
int got_keep = head_count(V);
/* 1. keep-count must match the reference exactly. The partial select and the old
* qsort keep the same NUMBER of tokens by construction (same cumulative-mass rule);
* a count divergence is a real bug, not a tie artifact. */
if (got_keep != ref_keep)
FAIL("keep-count mismatch: got %d, ref %d", got_keep, ref_keep);
/* 2. Detect ties across the WHOLE pre-truncation distribution, not just the kept set.
* A tie at the head/tail boundary makes which-side-a-token-lands-on interchangeable:
* both algorithms keep the right count but may keep different MEMBERS. So any input
* with a duplicated softmax value needs the relaxed multiset comparison below. We
* detect this on the reference softmax (pre-truncation) by sorting all V values. */
int has_ties = 0;
{
double *all = malloc((size_t)V * sizeof(double));
/* reconstruct the pre-truncation softmax the same way ref_build does */
double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i];
double s = 0, invt = 1.0 / (g_temp > 1e-4 ? g_temp : 1e-4);
for (int i = 0; i < V; i++){ all[i] = exp((lo[i]-mx)*invt); s += all[i]; }
for (int i = 0; i < V; i++) all[i] /= s;
for (int a = 1; a < V; a++){ double k = all[a]; int b = a-1;
while (b >= 0 && all[b] > k){ all[b+1] = all[b]; b--; } all[b+1] = k; }
for (int a = 1; a < V; a++) if (all[a] == all[a-1]){ has_ties = 1; break; }
free(all);
}
if (has_ties){
/* Multiset equality of the non-zero (head) values. Ties make membership
* interchangeable, so we compare sorted value-multisets, not id-aligned values.
* Tolerance is 1e-6 relative -- the engine uses float arithmetic, the reference
* double, so sub-ULP noise is expected (matches test_i4_grouped.c's convention). */
double *got = malloc((size_t)ref_keep * sizeof(double));
int gm = 0;
for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) got[gm++] = (double)g_pbuf[i];
if (gm != ref_keep)
FAIL("tie-shape head size mismatch: got %d non-zero, ref %d", gm, ref_keep);
for (int a = 1; a < gm; a++){ double k = got[a]; int b = a-1;
while (b >= 0 && got[b] > k){ got[b+1] = got[b]; b--; } got[b+1] = k; }
double *rsort = malloc((size_t)ref_keep * sizeof(double));
int rm = 0;
for (int i = 0; i < V; i++) if (ref[i] != 0.0) rsort[rm++] = ref[i];
for (int a = 1; a < rm; a++){ double k = rsort[a]; int b = a-1;
while (b >= 0 && rsort[b] > k){ rsort[b+1] = rsort[b]; b--; } rsort[b+1] = k; }
int mm = 0; double worst = 0;
for (int i = 0; i < gm; i++){
double d = fabs(got[i] - rsort[i]);
double rel = rsort[i] > 1e-30 ? d / rsort[i] : d;
if (rel > worst) worst = rel;
if (rel > 1e-6) mm++;
}
free(got); free(rsort);
if (mm) FAIL("tie-shape multiset mismatch: %d/%d head values differ beyond 1e-6 rel (worst %.3g)",
mm, ref_keep, worst);
} else {
/* No ties anywhere: membership is forced, so compare id-aligned head values. The
* engine computes in float (g_pbuf /= (float)s2) while the reference uses double,
* so the comparison is relative-tolerance (1e-6), not bit-exact -- the partial
* select and qsort accumulate s2 in the same descending order, so any difference
* is pure float-rounding noise, not an ordering bug. */
int bad = 0; int first_id = -1; float gv = 0, rv = 0; double worst = 0;
for (int i = 0; i < V; i++){
if (ref[i] == 0.0) continue; /* tail */
float want = (float)ref[i];
double d = fabs((double)g_pbuf[i] - (double)want);
double rel = fabs((double)want) > 1e-30 ? d / fabs((double)want) : d;
if (rel > worst) worst = rel;
if (rel > 1e-6){
bad++; if (first_id < 0){ first_id = i; gv = g_pbuf[i]; rv = want; }
if (bad > 3) break;
}
}
if (bad)
FAIL("head not within 1e-6 rel of reference: %d entries differ (first id %d: got %.9g want %.9g, worst %.3g)",
bad, first_id, (double)gv, (double)rv, worst);
}
/* 3. head must renormalize to 1.0 (within float epsilon) */
double sum = 0; for (int i = 0; i < V; i++) sum += g_pbuf[i];
if (fabs(sum - 1.0) > 1e-5)
FAIL("head does not sum to 1.0: sum=%.12g (keep=%d)", sum, got_keep);
free(ref); free(ridx);
printf(" ok [V=%d nuc=%.3f shape=%s keep=%d%s sum=%.10f]\n",
V, nuc, shape_name, got_keep, has_ties ? " (ties)" : "", sum);
}
#undef FAIL
/* deterministic xorshift32 RNG (matches the test_i4_grouped.c convention) */
static uint32_t rng_state = 0x12345678u;
static uint32_t xr(void){ rng_state ^= rng_state << 13; rng_state ^= rng_state >> 17;
rng_state ^= rng_state << 5; return rng_state; }
static double frand(void){ return (xr() >> 8) * (1.0 / 16777216.0); } /* [0,1) */
/* fill logits for a given shape. Shapes chosen to stress the comparator and the head/tail
* boundary differently. */
static void fill_shape(float *lo, int V, int shape){
switch (shape){
case 0: /* uniform -> every token equal probability -> massive tie plateau */
for (int i = 0; i < V; i++) lo[i] = 0.f; break;
case 1: /* peaked: one dominant token, rest small and distinct (no ties).
* The fixed hot-spots are clamped to V-1 so small V (incl. V=1) doesn't
* write out of bounds and corrupt heap metadata on the later free(lo). */
for (int i = 0; i < V; i++) lo[i] = (float)(-1.0 - frand()*4.0);
lo[0] = 3.f; lo[V/3<V?V/3:V-1] = 1.f; lo[V/2<V?V/2:V-1] = 0.5f; break;
case 2: /* all-equal distinct decay (no ties): geometric, strictly decreasing */
for (int i = 0; i < V; i++) lo[i] = (float)(-0.001 * (double)i); break;
case 3: /* plateau ties: blocks of equal value -> comparator tie handling */
for (int i = 0; i < V; i++) lo[i] = (float)(-(double)(i / 7)); /* 7-wide plateaus */
break;
case 4: /* sharp-tail: a few hot, then a long flat floor (small tie at the floor).
* Hot count is min(12,V) so V<12 (incl. V=1) stays in bounds. */
for (int i = 0; i < V; i++) lo[i] = -8.f;
{ int hot = V<12 ? V : 12; for (int i = 0; i < hot; i++) lo[i] = (float)(2.0 - frand()); } break;
}
}
int main(void){
/* sizes: small for exhaustive tie detection up to near-production scale */
int sizes[] = {1, 2, 8, 64, 257, 1519}; /* 1519 ~= V/100 of GLM-5.2 */
double nucs[] = {0.001, 0.5, 0.9, 0.999}; /* tight -> almost-everything */
int n_shapes = 5;
/* temperature used by dist_build: pick a normal serving value */
g_temp = 0.7f;
int cases = 0;
for (size_t si = 0; si < sizeof(sizes)/sizeof(sizes[0]); si++){
int V = sizes[si];
/* dist_build allocates g_pbuf/g_pidx ONCE and reuses them (single-V invariant in
* real serving, where V is the constant model vocab). This sweep varies V, so free
* and force a reallocation per size -- otherwise a later, larger V would overflow
* the buffer sized for the first (smallest) V. */
free(g_pbuf); g_pbuf = NULL; free(g_pidx); g_pidx = NULL;
float *lo = malloc((size_t)V * sizeof(float));
for (int shape = 0; shape < n_shapes; shape++){
fill_shape(lo, V, shape);
for (size_t ni = 0; ni < sizeof(nucs)/sizeof(nucs[0]); ni++){
char label[32]; snprintf(label, sizeof(label), "size[%zu]/shape[%d]", si, shape);
const char *sn = (const char*[]){"uniform","peaked","geometric","plateau","sharptail"}[shape];
check_case(label, V, nucs[ni], sn, lo);
cases++;
}
}
free(lo);
}
/* guard-off path: g_nuc >= 1 must skip truncation entirely (full softmax kept) */
{
int V = 256; float lo[256];
for (int i = 0; i < V; i++) lo[i] = (float)(frand()*4 - 2);
g_nuc = 1.0f; dist_build(lo, V);
int nz = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) nz++;
if (nz != V){ fprintf(stderr, " FAIL [guard-off nuc=1.0]: %d/%d entries kept, expected all\n", nz, V); g_nfails++; }
else printf(" ok [guard-off nuc=1.0 keep=%d]\n", nz);
cases++;
g_nuc = 0.0f; dist_build(lo, V);
nz = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) nz++;
if (nz != V){ fprintf(stderr, " FAIL [guard-off nuc=0.0]: %d/%d entries kept, expected all\n", nz, V); g_nfails++; }
else printf(" ok [guard-off nuc=0.0 keep=%d]\n", nz);
cases++;
}
/* extreme tie edge case: V=1, single token -> keep=1 regardless of nuc */
{
float lo[1] = {5.f};
g_nuc = 0.5f; dist_build(lo, 1);
if (g_pbuf[0] == 0.f || !(fabs((double)g_pbuf[0] - 1.0) < 1e-6)){
fprintf(stderr, " FAIL [V=1]: g_pbuf[0]=%.9g, expected 1.0\n", (double)g_pbuf[0]); g_nfails++;
} else printf(" ok [V=1 keep=1]\n");
cases++;
}
printf("\ntest_topp: %d cases run, %d failure(s)\n", cases, g_nfails);
if (g_nfails){ printf("test_topp: FAIL\n"); return 1; }
printf("test_topp: ok\n");
return 0;
}