Merge pull request #224 from michael-denyer/fix-kv-alloc-double-free
glm: fix kv_alloc double-free on KV re-allocation (stale pre-Metal free block)
This commit is contained in:
+4
-1
@@ -146,7 +146,7 @@ else
|
||||
PYTHON ?= python3
|
||||
endif
|
||||
CUDA_OBJ =
|
||||
TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE)
|
||||
TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_kv_alloc$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE)
|
||||
|
||||
# Windows CUDA DLL path: host links the loader, NOT cudart.
|
||||
ifneq ($(IS_WIN),)
|
||||
@@ -260,6 +260,9 @@ tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h
|
||||
tests/test_idot$(EXE): tests/test_idot.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
|
||||
@@ -3066,7 +3066,6 @@ static void kv_alloc(Model *m, int max_t){
|
||||
m->kv_dev_valid[i]=0;
|
||||
}
|
||||
#endif
|
||||
if(k->Lc){ for(int i=0;i<c->n_layers+1;i++){ free(k->Lc[i]); free(k->Rc[i]); } free(k->Lc); free(k->Rc); }
|
||||
if(k->Lc){ for(int i=0;i<c->n_layers+1;i++){
|
||||
#ifdef COLI_METAL
|
||||
if(g_metal_enabled){ coli_metal_unregister(k->Lc[i]); coli_metal_unregister(k->Rc[i]); }
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
/* kv_alloc must survive re-allocation on the same KVState: every free path is
|
||||
* guarded by if(k->Lc) precisely so callers (context resize, slot re-init) can
|
||||
* call it again. A stale duplicate free block frees every Lc[i]/Rc[i] and both
|
||||
* arrays twice on the second call -> allocator abort. No model file needed:
|
||||
* the CPU path of kv_alloc only reads c->n_layers/kv_lora/qk_rope. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#undef main
|
||||
|
||||
int main(void){
|
||||
static Model m;
|
||||
m.c.n_layers=2; m.c.kv_lora=8; m.c.qk_rope=4;
|
||||
m.kv=calloc(1,sizeof(KVState));
|
||||
kv_alloc(&m,16);
|
||||
for(int i=0;i<m.c.n_layers+1;i++){ m.Lc[i][0]=1.0f; m.Rc[i][0]=1.0f; }
|
||||
kv_alloc(&m,32); /* the re-allocation path under test */
|
||||
for(int i=0;i<m.c.n_layers+1;i++){
|
||||
m.Lc[i][(int64_t)32*m.c.kv_lora-1]=2.0f;
|
||||
m.Rc[i][(int64_t)32*m.c.qk_rope-1]=2.0f;
|
||||
}
|
||||
printf("OK kv_alloc re-allocation\n");
|
||||
return 0;
|
||||
}
|
||||
Reference in New Issue
Block a user