fix(rebase): adapt fmt=4 to dev's quant.h refactor + colibri.c rename
Dev refactored glm.c → colibri.c and extracted the matmul/quant kernels into quant.h (matmul, matmul_q, matmul_i4, matmul_i4_grouped, matmul_i2, quant_scratch, dot_i4i8, matmul_q_idot, matmul_i4_idot, etc. all live there now). The original #298 commit re-added all of these inline; on rebase they became duplicate definitions. Resolution: - Removed the ~700-line duplicate block (everything dev moved to quant.h) - Kept ONLY the unique fmt=4 contribution: matmul_i4_grouped_pair (the fused gate+up kernel that reads x once instead of twice, ~33% decode speedup) + the fmt=4 branch in expert_gate_up that dispatches to it. Dev's expert_gate_up only fused fmt==2; this adds the fmt==4 case. - Forward-declared matmul_i4_grouped_pair before expert_gate_up. - Fixed quant_matmul call site in the ragged attention path (backend_cuda.cu) to pass gs/ng — the kernel signature gained those args in the attention scales fix, but dev's new ragged path called it with the old signature. Build-verified: colibri.exe (CPU + COLI_CUDA) and coli_cuda.dll both compile clean on the rebased branch.
This commit is contained in:
+1
-1
@@ -1122,7 +1122,7 @@ extern "C" int coli_cuda_attention_project_ragged(ColiCudaTensor *w,ColiCudaTens
|
||||
attention_absorb_ragged_kernel<<<dim3(H,S),256,shared,dc->stream>>>(dc->ac,dc->aq,ddl,ddr,
|
||||
dn,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
quant_matmul<<<dim3(proj->O,S),256,0,dc->stream>>>(dc->y,dc->ac,proj->weights,
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I));
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I),proj->gs,proj->ng);
|
||||
return cuda_ok(cudaGetLastError(),"ragged attention launch")&&
|
||||
cuda_ok(cudaMemcpyAsync(out,dc->y,ob,cudaMemcpyDeviceToHost,dc->stream),"ragged output download")&&
|
||||
cuda_ok(cudaStreamSynchronize(dc->stream),"ragged attention synchronize");
|
||||
|
||||
Reference in New Issue
Block a user