diff --git a/c/Makefile b/c/Makefile index de8313c..1e0b1ea 100644 --- a/c/Makefile +++ b/c/Makefile @@ -146,7 +146,7 @@ else PYTHON ?= python3 endif CUDA_OBJ = -TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) +TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_kv_alloc$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) # Windows CUDA DLL path: host links the loader, NOT cudart. ifneq ($(IS_WIN),) @@ -260,6 +260,9 @@ tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h tests/test_idot$(EXE): tests/test_idot.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) diff --git a/c/tests/test_kv_alloc.c b/c/tests/test_kv_alloc.c new file mode 100644 index 0000000..40ec00d --- /dev/null +++ b/c/tests/test_kv_alloc.c @@ -0,0 +1,23 @@ +/* kv_alloc must survive re-allocation on the same KVState: every free path is + * guarded by if(k->Lc) precisely so callers (context resize, slot re-init) can + * call it again. A stale duplicate free block frees every Lc[i]/Rc[i] and both + * arrays twice on the second call -> allocator abort. No model file needed: + * the CPU path of kv_alloc only reads c->n_layers/kv_lora/qk_rope. */ +#define main coli_glm_main_unused +#include "../glm.c" +#undef main + +int main(void){ + static Model m; + m.c.n_layers=2; m.c.kv_lora=8; m.c.qk_rope=4; + m.kv=calloc(1,sizeof(KVState)); + kv_alloc(&m,16); + for(int i=0;i