Compare commits
152 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4aca05900f | |||
| eaa68f4005 | |||
| 1b8f3076b6 | |||
| ac04b37af9 | |||
| e16ec68168 | |||
| 4cc9885cb3 | |||
| f939191404 | |||
| 0d2fb6f8a2 | |||
| cbbe31094a | |||
| 68ac9ff696 | |||
| 507c8a1808 | |||
| 87cd0aafed | |||
| 880cfb4ec5 | |||
| 133d271bbf | |||
| c98e5f8bf8 | |||
| 7291fd19ac | |||
| 8b69c89f26 | |||
| f4409fd7ab | |||
| d96e4254a4 | |||
| bfb80000be | |||
| 6e29260635 | |||
| 86987154f5 | |||
| 1a260a4605 | |||
| f42adfae76 | |||
| 932f3678b0 | |||
| 5e42e70ec3 | |||
| cccf8ec5cb | |||
| 84514a5f39 | |||
| d7b855f43f | |||
| d7211632b4 | |||
| 49c11539b2 | |||
| ebd85a781b | |||
| 31526ae619 | |||
| e48513c1c6 | |||
| 93cde63cd3 | |||
| c98b451738 | |||
| 04ed0049a4 | |||
| 73badea697 | |||
| aa825d361a | |||
| 06961728d8 | |||
| e78fcfcbca | |||
| 78a773e539 | |||
| 1aacfcff01 | |||
| e809171e62 | |||
| d34461c8d8 | |||
| d3362303f5 | |||
| ff38be0fee | |||
| 48d4af691d | |||
| f8e0612f26 | |||
| b347092e5c | |||
| 85e7dffb85 | |||
| 7de49fa02d | |||
| 505a0f69aa | |||
| 0337697023 | |||
| af23f314fd | |||
| 1765ed80ac | |||
| bb2bae8904 | |||
| 3ddbc655ef | |||
| 3455eeea0a | |||
| 0b05f6a76b | |||
| 171aa69fcb | |||
| bd3efb1c97 | |||
| f27a89ea82 | |||
| cf126cbcf9 | |||
| d43b54534f | |||
| e48346162a | |||
| 42a5417c13 | |||
| ae4e31a15a | |||
| 57e67d6c2f | |||
| 536d8bfd1a | |||
| a74e3e0c3a | |||
| e9b36141a4 | |||
| 9c5ab39b62 | |||
| 4b704823db | |||
| 9f30916003 | |||
| 85107f8ce6 | |||
| 453d1401ba | |||
| c4b2cf6db0 | |||
| 9c698b29a6 | |||
| 17fbdfa8f8 | |||
| a7ef058fc0 | |||
| 5c84b25645 | |||
| fbabaa4544 | |||
| 1f00142b25 | |||
| 26bd8b403a | |||
| 70fb5b00f3 | |||
| 34f6e50091 | |||
| 113ece3bc7 | |||
| e135119f19 | |||
| 042fd64a7c | |||
| 8efa9ec6c3 | |||
| e011092ce1 | |||
| 82d06000b5 | |||
| 309f20c939 | |||
| 0049ea15ae | |||
| 22560ad15b | |||
| bca4e95d92 | |||
| 3e45074bf2 | |||
| 7d01d21023 | |||
| 1d9f554715 | |||
| ab55f4900c | |||
| 4586d33c60 | |||
| e9225d2ffc | |||
| 63966ba3f8 | |||
| 26d6b2d662 | |||
| 468b190db9 | |||
| 61004dcb84 | |||
| 8a9a0fca4d | |||
| 083fda5b0a | |||
| bc69a9a6d0 | |||
| 93b4a8e78e | |||
| e486574442 | |||
| 420a0720c3 | |||
| f853ea8a0b | |||
| 8f33bd153b | |||
| cfcc742591 | |||
| ebc851edb3 | |||
| 40a3596354 | |||
| 845af6378d | |||
| 4775f28ceb | |||
| 51fe03f615 | |||
| da97f0dbf1 | |||
| 58fd0e557f | |||
| bcb984a61f | |||
| e93d574e13 | |||
| a6d0a6c4a7 | |||
| 73aef4d010 | |||
| ac39be6b62 | |||
| 3ffe4bb75e | |||
| 72e36772f5 | |||
| fae4b2cc3e | |||
| 22509fccde | |||
| ca39e5333f | |||
| 741d46ba25 | |||
| d86a6b93ad | |||
| 92ddc2234f | |||
| dc196633f1 | |||
| 288edd7190 | |||
| 5ad4d540ab | |||
| 9e2d41567c | |||
| 364b9741e2 | |||
| c769e04d13 | |||
| b6bae91b66 | |||
| 2d8d2951ee | |||
| 1ac2e7b487 | |||
| 6ade4093de | |||
| ca788833ab | |||
| 24058d3de8 | |||
| b3fdb145f8 | |||
| 4ae4f61d16 | |||
| c4624276f3 | |||
| ac1f7a8f38 |
@@ -0,0 +1,10 @@
|
||||
BasedOnStyle: LLVM
|
||||
IndentWidth: 4
|
||||
ColumnLimit: 120
|
||||
AllowShortFunctionsOnASingleLine: All
|
||||
AllowShortIfStatementsOnASingleLine: AllIfsAndElse
|
||||
AllowShortLoopsOnASingleLine: true
|
||||
BreakBeforeBraces: Attach
|
||||
PointerAlignment: Right
|
||||
SpaceAfterCStyleCast: false
|
||||
SortIncludes: false
|
||||
@@ -0,0 +1,24 @@
|
||||
root = true
|
||||
|
||||
[*]
|
||||
charset = utf-8
|
||||
end_of_line = lf
|
||||
insert_final_newline = true
|
||||
trim_trailing_whitespace = true
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
|
||||
[*.{c,h,cu}]
|
||||
indent_size = 4
|
||||
|
||||
[*.{ts,tsx,js,json,css}]
|
||||
indent_size = 2
|
||||
|
||||
[Makefile]
|
||||
indent_style = tab
|
||||
|
||||
[*.yml]
|
||||
indent_size = 2
|
||||
|
||||
[*.md]
|
||||
trim_trailing_whitespace = false
|
||||
@@ -12,8 +12,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Build glm
|
||||
run: cd c && make glm
|
||||
- name: Build colibri
|
||||
run: cd c && make colibri
|
||||
- name: C test suite
|
||||
run: cd c && make test-c
|
||||
|
||||
@@ -105,13 +105,13 @@ jobs:
|
||||
make cuda-dll CUDA_ARCH=sm_80
|
||||
test -f coli_cuda.dll || { echo "cuda-dll reported success but produced no DLL" >&2; exit 1; }
|
||||
echo "coli_cuda.dll built (MSVC host)"
|
||||
- name: make glm CUDA_DLL=1 (host links backend_loader, not cudart)
|
||||
- name: make colibri CUDA_DLL=1 (host links backend_loader, not cudart)
|
||||
shell: msys2 {0}
|
||||
run: |
|
||||
cd c
|
||||
make glm CUDA_DLL=1
|
||||
test -f glm.exe || { echo "glm CUDA_DLL=1 reported success but produced no exe" >&2; exit 1; }
|
||||
echo "glm.exe built against the DLL loader"
|
||||
make colibri CUDA_DLL=1
|
||||
test -f colibri.exe || { echo "colibri CUDA_DLL=1 reported success but produced no exe" >&2; exit 1; }
|
||||
echo "colibri.exe built against the DLL loader"
|
||||
|
||||
web:
|
||||
name: Web UI
|
||||
|
||||
@@ -10,6 +10,8 @@ desktop/src-tauri/target/
|
||||
desktop/src-tauri/gen/
|
||||
|
||||
# binari compilati (si rigenerano con make / coli build)
|
||||
c/colibri
|
||||
c/colibri.exe
|
||||
c/glm
|
||||
c/glm.exe
|
||||
c/olmoe
|
||||
@@ -35,6 +37,7 @@ c/tests/test_schema_gbnf
|
||||
c/tests/test_schema_gbnf.exe
|
||||
c/tests/test_compat_direct
|
||||
c/tests/test_compat_direct.exe
|
||||
result
|
||||
|
||||
# oracoli tiny generati (make_glm_oracle.py) e dati benchmark scaricati
|
||||
c/glm_tiny/
|
||||
@@ -68,3 +71,7 @@ c/tests/test_decode_batch
|
||||
c/tests/test_i4_acc512
|
||||
c/tests/test_idot
|
||||
c/tests/test_uring
|
||||
olmoe_merged/
|
||||
olmoe_i4/
|
||||
c/olmoe_merged/
|
||||
c/olmoe_i4/
|
||||
|
||||
+260
@@ -0,0 +1,260 @@
|
||||
<p align="center">
|
||||
<img src="assets/colibri.svg" width="500" alt="colibrì — motore piccolo, modello immenso">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a> · <a href="README.zh-TW.md">繁體中文</a> · Italiano
|
||||
</p>
|
||||
|
||||
**Motore piccolo, modello immenso.** Esegui **GLM-5.2 (744 miliardi di parametri, MoE)** su un computer consumer con ~25 GB di RAM — in C puro, zero dipendenze, caricando gli expert dal disco in streaming.
|
||||
|
||||
Colibrì è un runtime MoE leggero e che preserva la qualità: tratta VRAM, RAM e
|
||||
disco come un'unica gerarchia di memoria gestita. Se la memoria veloce non basta
|
||||
il modello rallenta, ma la policy predefinita **non cambia mai silenziosamente la
|
||||
precisione del modello né la semantica del router**.
|
||||
|
||||
```
|
||||
$ ./coli chat
|
||||
🐦 colibrì v1.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
|
||||
✓ ready in 32s · resident 9.9 GB
|
||||
› ciao!
|
||||
◆ Ciao! 😊 Come posso aiutarti oggi?
|
||||
```
|
||||
|
||||
## Guardalo in azione
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-dashboard.png" width="900" alt="dashboard web di colibrì — metriche live, pannello hardware, livelli degli expert">
|
||||
</p>
|
||||
<p align="center"><em>La dashboard web (<code>./coli web</code>): un modello da 744B a <strong>4 tok/s, TTFT 1.6 s, disco 0</strong> —
|
||||
residenza completa degli expert su 6× RTX 5090, con metriche token in tempo reale, breakdown dei tempi per turno,
|
||||
la barra dei livelli VRAM/RAM/disco e il mini-cervello live nell'angolo.</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-brain.png" width="900" alt="la pagina Brain — 19.456 expert come una corteccia vivente">
|
||||
</p>
|
||||
<p align="center"><em>La pagina <strong>Brain</strong>: tutti i 19.456 expert come una corteccia vivente — il colore indica
|
||||
il livello di archiviazione, la luminosità il calore di routing, e ogni expert instradato in un turno
|
||||
lampeggia bianco. Passando il cursore si vede l'<a href="https://github.com/JustVugg/colibri/issues/175">affinità
|
||||
tematica misurata</a> dell'expert.</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-atlas.png" width="900" alt="la pagina Atlas — l'atlante misurato degli expert come una galassia 3D">
|
||||
</p>
|
||||
<p align="center"><em>La pagina <strong>Atlas</strong>: l'<a href="https://github.com/JustVugg/colibri/issues/175">atlante
|
||||
misurato degli expert</a> come una galassia 3D — 13.260 expert caratterizzati, 1.041 specialisti
|
||||
replicabili che si raggruppano per argomento (poesia, legge, cinese, SQL…). La posizione deriva
|
||||
dall'affinità di routing misurata, non da un embedding appreso. Trascinare per ruotare.</em></p>
|
||||
|
||||
## L'idea
|
||||
|
||||
Un modello Mixture-of-Experts da 744B attiva solo ~40B parametri per token — e
|
||||
solo ~11 GB di quelli cambiano da un token all'altro (gli expert instradati):
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/sparse.png" width="880" alt="solo ~5.4% dei parametri è attivo per token">
|
||||
</p>
|
||||
|
||||
Il modello non ha bisogno di *stare* in memoria veloce — ha bisogno di essere
|
||||
**piazzato**:
|
||||
|
||||
- la **parte densa** (attenzione, expert condivisi, embedding — ~17B parametri)
|
||||
resta **residente in RAM a int4** (~9.9 GB);
|
||||
- i **19.456 expert instradati** (75 layer MoE × 256 + la testa MTP, ~19 MB
|
||||
ciascuno a int4) stanno **su disco** (~370 GB) e vengono **caricati on demand
|
||||
in streaming**, con una cache LRU per layer, un hot-store pinnato che impara,
|
||||
e un livello VRAM opzionale.
|
||||
|
||||
Il motore è un singolo file C (`c/colibri.c`) più header piccoli. Niente BLAS,
|
||||
niente Python a runtime, niente GPU obbligatoria.
|
||||
|
||||
## Come funziona
|
||||
|
||||
### Il percorso di ogni token
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/token-path.png" width="880" alt="instrada → unione → piazza → sovrapponi → impara">
|
||||
</p>
|
||||
|
||||
Ogni layer di ogni token percorre gli stessi cinque passi. L'obiettivo
|
||||
progettuale è che **il piazzamento decide solo la velocità** — le decisioni
|
||||
del router e la precisione dei pesi sono identiche sia che un expert risponda
|
||||
dalla VRAM sia dal disco.
|
||||
|
||||
### Una gerarchia di memoria, non un requisito di memoria
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/tiers.png" width="880" alt="residenza expert a tre livelli: VRAM / RAM / NVMe">
|
||||
</p>
|
||||
|
||||
Lo stesso motore copre l'intero spettro: su un portatile da 25 GB tutto viene
|
||||
caricato dal disco in streaming (lento, ma corretto); su un host grande l'intero
|
||||
set di expert diventa residente (`CUDA_EXPERT_GB=auto PIN_GB=all`) e il disco
|
||||
esce completamente dal percorso di decode. Tra i livelli c'è una **cache che
|
||||
impara**: il motore registra quali expert il *tuo* carico di lavoro instrada
|
||||
(`.coli_usage`, aggiornato a ogni turno) e fissa automaticamente i più caldi —
|
||||
colibrì diventa letteralmente più veloce man mano che lo usi. Sugli host
|
||||
multi-socket, `COLI_NUMA=1` interlaccia i pesi residenti tra i controller di
|
||||
memoria ([#82](https://github.com/JustVugg/colibri/issues/82)).
|
||||
|
||||
### Mai aspettare il disco due volte
|
||||
|
||||
I miss nella cache costano caro, quindi il motore investe la maggior parte
|
||||
della sua astuzia per evitarli e sovrapporli: le tre matrici di ogni expert sono
|
||||
memorizzate contigue e lette con un unico `pread`; un pool I/O asincrono
|
||||
limitato (`PIPE=1`, attivo per default) carica gli expert mancanti mentre quelli
|
||||
residenti calcolano; le posizioni in batch leggono ogni expert unico una sola
|
||||
volta (**batch-union**); un thread di lookahead del router (`PILOT=1`) fa il
|
||||
prefetch degli expert del layer successivo — il routing è misurabilmente
|
||||
**prevedibile al 71.6% un layer in anticipo**. Sulle GPU, la pipeline residente
|
||||
(`COLI_CUDA_PIPE=2`) mantiene il flusso residuo on-device tra i layer, così il
|
||||
loop CPU degli expert procede senza interruzioni; su Apple Silicon un backend
|
||||
[Metal](docs/metal.md) sperimentale esegue la matmul batch degli expert sulla
|
||||
GPU a memoria unificata.
|
||||
|
||||
### Modello fedele, stato compresso
|
||||
|
||||
Il forward pass è validato **token-esatto contro un oracle `transformers`**
|
||||
(teacher-forcing 32/32). L'attenzione MLA memorizza uno stato KV compresso — 576
|
||||
float/token invece di 32.768 (**57× più piccolo**) — e lo persiste tra i
|
||||
riavvii (`.coli_kv`): le conversazioni riaprono "calde", senza alcun re-prefill,
|
||||
byte-identiche a una sessione ininterrotta. L'attenzione sparsa DSA (il
|
||||
lightning indexer di GLM-5.2) è implementata fedelmente e validata forzando la
|
||||
selezione di tutte le chiavi per riprodurre esattamente l'attenzione densa.
|
||||
|
||||
### Decodifica speculativa, onestamente
|
||||
|
||||
La testa MTP nativa di GLM-5.2 propone token che il modello principale verifica
|
||||
in un unico forward batch — 2.2–2.8 token/forward quando conviene. Due regole
|
||||
conquistate a caro prezzo sono i default: la testa MTP deve essere **int8** (le
|
||||
teste int4 crollano al 0–4% di accettazione,
|
||||
[#8](https://github.com/JustVugg/colibri/issues/8)), e draft e verifica devono
|
||||
calcolare **la stessa funzione** — `SPEC_PIN=1` fissa entrambi sulla stessa
|
||||
famiglia di kernel ([#163](https://github.com/JustVugg/colibri/issues/163)
|
||||
contiene l'intera indagine forense). I draft forzati da grammatica
|
||||
([`GRAMMAR=file.gbnf`](docs/grammar-draft.md)) aggiungono accettazione quasi
|
||||
gratuita sull'output JSON vincolato. Se la speculazione conviene dipende dalla
|
||||
temperatura della cache — misura, e usa `DRAFT=0` quando non paga.
|
||||
|
||||
## Cosa ottiene
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/ladder.png" width="880" alt="velocità di decode misurata per classe hardware">
|
||||
</p>
|
||||
|
||||
Stesso motore, stesso container int4 — cambia solo dove risiedono gli expert.
|
||||
Punti salienti dalle [tabelle benchmark complete](docs/benchmarks.md):
|
||||
|
||||
- **6× RTX 5090, residenza completa:** 5.8–6.8 tok/s in decode, TTFT ~13 s
|
||||
([log dell'esperimento](docs/experiments/glm52-6x5090-2026-07-12.md));
|
||||
- **desktop solo-CPU da 128 GB:** ~1.8 tok/s a cache calda
|
||||
([#200](https://github.com/JustVugg/colibri/issues/200));
|
||||
- **singola RTX 5070 Ti, classe laptop:** 1.07 tok/s tramite la pipeline
|
||||
GPU-residente ([#273](https://github.com/JustVugg/colibri/issues/273));
|
||||
- **macchina di sviluppo da 25 GB:** 0.05–0.1 tok/s a freddo — il punto di
|
||||
partenza dimostrato da cui è nato il progetto, e ancora oggi la baseline onesta.
|
||||
|
||||
La qualità è misurata, non presunta: il costo di quantizzazione del container
|
||||
int4 e le ablazioni su granularità delle scale e rotazione sono in
|
||||
[docs/benchmarks.md](docs/benchmarks.md#quality-benchmark) e
|
||||
[#108](https://github.com/JustVugg/colibri/issues/108)/[#81](https://github.com/JustVugg/colibri/issues/81).
|
||||
|
||||
## Per iniziare
|
||||
|
||||
### 1. Scarica il modello
|
||||
|
||||
Un container **GLM-5.2 int4** pre-convertito è su Hugging Face — **usa la
|
||||
versione con le teste MTP int8**:
|
||||
|
||||
**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp**
|
||||
|
||||
> ⚠️ Il mirror originale contiene teste MTP int4 → accettazione dei draft allo 0%
|
||||
> ([#8](https://github.com/JustVugg/colibri/issues/8)). Verifica la tua versione:
|
||||
> `ls -l <modello>/out-mtp-*` — int8 (corretto) è `3527131672 / 5366238584 / 1065950496`.
|
||||
|
||||
Oppure converti tu stesso dalla sorgente FP8 — un unico comando riprendibile che
|
||||
non richiede mai i 756 GB completi su disco contemporaneamente:
|
||||
|
||||
```bash
|
||||
cd c && ./setup.sh # verifica gcc/OpenMP, compila, autotest
|
||||
./coli convert --model /nvme/glm52_i4 # scarica e converti shard per shard (python, una tantum)
|
||||
```
|
||||
|
||||
### 2. Esegui
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat # budget RAM, cache e MTP rilevati automaticamente
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli plan # mostra il piazzamento pianificato VRAM/RAM/disco
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli doctor # controllo di idoneità (sola lettura)
|
||||
./coli web --model /nvme/glm52_i4 # API + dashboard web sulla stessa porta
|
||||
./coli serve --model /nvme/glm52_i4 # solo API compatibile OpenAI
|
||||
```
|
||||
|
||||
Il motore a runtime è puro C — python si usa solo per il convertitore (una tantum)
|
||||
e per il gateway API opzionale.
|
||||
|
||||
### 3. Approfondisci
|
||||
|
||||
| argomento | documento |
|
||||
|---|---|
|
||||
| Benchmark, dati dalla comunità, misurazioni di qualità | [docs/benchmarks.md](docs/benchmarks.md) |
|
||||
| Parametri di tuning, policy, cache che impara, prefetch | [docs/tuning.md](docs/tuning.md) |
|
||||
| Build nativa su Windows 11 (con CUDA DLL) | [docs/windows.md](docs/windows.md) |
|
||||
| Backend CUDA, livello expert in VRAM, residenza completa | [docs/cuda.md](docs/cuda.md) |
|
||||
| Backend Metal per Apple Silicon | [docs/metal.md](docs/metal.md) |
|
||||
| API compatibile OpenAI, KV slot, dashboard web | [docs/api.md](docs/api.md) |
|
||||
| Draft forzati da grammatica (output strutturato) | [docs/grammar-draft.md](docs/grammar-draft.md) |
|
||||
| Inventario delle variabili d'ambiente | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
|
||||
|
||||
## Sostenere il progetto
|
||||
|
||||
colibrì è nato come progetto di una sola persona su un portatile con 12 core
|
||||
e 25 GB di RAM; oggi i suoi numeri arrivano da una comunità di macchine reali.
|
||||
Se ti è utile:
|
||||
|
||||
- ⭐ metti una stella al repository e condividilo;
|
||||
- 🐛 apri issue con i numeri di benchmark del tuo hardware — i datapoint
|
||||
fanno avanzare questo progetto più di qualsiasi altra cosa;
|
||||
- 💬 contattaci via GitHub issues per sponsorizzare lo sviluppo o donare hardware.
|
||||
|
||||
## Struttura del repository
|
||||
|
||||
```
|
||||
Makefile punto d'ingresso root per build/check
|
||||
c/
|
||||
├── colibri.c motore principale
|
||||
├── quant.h kernel matmul quantizzati (SIMD multi-architettura)
|
||||
├── sample.h campionamento, RNG, set di stop
|
||||
├── kv_persist.h persistenza KV su disco (.coli_kv)
|
||||
├── telemetry.h protocollo dashboard, statistiche, usage
|
||||
├── st.h, tok.h, json.h header di runtime
|
||||
├── backend_cuda.* livello CUDA opzionale
|
||||
├── Makefile build e check locali
|
||||
├── coli CLI utente
|
||||
├── openai_server.py gateway HTTP compatibile OpenAI
|
||||
├── setup.sh setup locale in un solo comando
|
||||
├── tools/ conversione offline, fixture e benchmark
|
||||
├── scripts/ helper per conversioni lunghe
|
||||
└── tests/ test C e Python senza dipendenze
|
||||
web/ UI browser (puro client API OpenAI)
|
||||
desktop/ shell desktop Tauri v2 che racchiude la web UI
|
||||
docs/ documentazione di riferimento, esperimenti, media
|
||||
```
|
||||
|
||||
Il percorso a runtime resta intenzionalmente piatto e leggibile: `colibri.c`
|
||||
più i suoi header. Dalla radice del repository, `make`, `make check` e
|
||||
`make clean` delegano al Makefile del motore.
|
||||
|
||||
## Perché "colibrì"
|
||||
|
||||
Il colibrì pesa pochi grammi, sta sospeso nel vuoto e visita un migliaio di
|
||||
fiori al giorno. Questo motore tiene in vita un gigante da 744 miliardi di
|
||||
parametri con le razioni di un colibrì: 25 GB di RAM, dodici core CPU e
|
||||
tanta pazienza col disco.
|
||||
|
||||
Il nome è rimasto in italiano perché questa è la lingua in cui è stato scritto
|
||||
il primo prototipo — i commenti nel codice lo testimoniano ancora.
|
||||
|
||||
## Licenza
|
||||
|
||||
Apache 2.0. I pesi di GLM-5.2 sono rilasciati da Z.ai sotto licenza MIT.
|
||||
@@ -9,7 +9,7 @@
|
||||
|
||||
<p align="center">
|
||||
<a href="https://justvugg.github.io/colibri"><b>Website</b></a> ·
|
||||
English · <a href="README.zh-TW.md">繁體中文</a>
|
||||
English · <a href="README.zh-CN.md">简体中文</a> · <a href="README.zh-TW.md">繁體中文</a> · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**Tiny engine, immense model.** Run **GLM-5.2 (744B-parameter MoE)** on a consumer machine with ~25 GB of RAM — in pure C, with zero dependencies, by streaming experts from disk.
|
||||
@@ -50,6 +50,16 @@ brightness is routing heat, and every expert routed in a turn flashes white. Hov
|
||||
as a 3-D galaxy — 13,260 characterised experts, 1,041 replicated specialists clustering by topic
|
||||
(poetry, law, Chinese, SQL…). Position is measured routing affinity, not a learned embedding. Drag to spin.</em></p>
|
||||
|
||||
## The vision
|
||||
|
||||
Frontier models should not be sealed inside datacenters. colibrì exists so that
|
||||
**anyone curious enough can open one up**: run a 744B-parameter mind on hardware
|
||||
you already own, watch every expert fire in real time, and change the code that
|
||||
does it. Not renting intelligence behind an API — *holding* it: probing it,
|
||||
measuring it, improving it. Every optimisation in this project started with
|
||||
someone measuring something on their own machine; the engine is deliberately
|
||||
small enough that the next one can come from you.
|
||||
|
||||
## The idea
|
||||
|
||||
A 744B Mixture-of-Experts model activates only ~40B parameters per token — and
|
||||
@@ -67,6 +77,18 @@ So the model doesn't need to *fit* in fast memory — it needs to be **placed**:
|
||||
at int4) live **on disk** (~370 GB) and are **streamed on demand**, with a
|
||||
per-layer LRU cache, a learned pinned hot-store, and an optional VRAM tier.
|
||||
|
||||
Think of the core algorithm as **a JIT, but for weights**. A compiler JIT never
|
||||
compiles the whole program — it watches what actually runs and compiles the hot
|
||||
paths, just in time. colibrì makes the same bet about a 744B parameter space:
|
||||
parameters are not resident state to be held, they are **data to be staged**
|
||||
across a heterogeneous storage hierarchy (VRAM / RAM / NVMe), exactly when the
|
||||
router proves they are needed. Measured routing heat decides which experts earn
|
||||
which tier, the router runs a layer ahead so prefetch hides the staging latency,
|
||||
and — like a JIT — the engine learns your workload: the more you run, the hotter
|
||||
the right experts get. It works because routing has measurable structure (see
|
||||
the [expert atlas](https://github.com/JustVugg/colibri/issues/175)) — and
|
||||
structure is cacheable.
|
||||
|
||||
The engine is a single C file (`c/glm.c`) plus small headers. No BLAS, no Python
|
||||
at runtime, no GPU required.
|
||||
|
||||
@@ -88,6 +110,22 @@ precision are the same whether an expert answered from VRAM or from disk.
|
||||
<img src="docs/media/tiers.png" width="880" alt="VRAM / RAM / NVMe three-tier expert residency">
|
||||
</p>
|
||||
|
||||
### Dual-SSD: two copies of the model, twice the read bandwidth
|
||||
|
||||
Decode is disk-bound on most machines, and expert reads are read-only — so if you have a **second SSD**, put a full copy of the model on it and let the engine stream from both drives at once:
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/fast/glm52_i4 COLI_MODEL_MIRROR=/second/glm52_i4 ./coli chat
|
||||
COLI_DISK_WEIGHTS=9,3 ... # optional: primary,mirror bandwidth ratio (else measured at startup)
|
||||
```
|
||||
|
||||
Each expert is routed to one drive by a deterministic hash, weighted by the two drives' measured (or declared) bandwidth, so readahead/PILOT prefetch and the demand read always hit the same drive and nothing is cached twice. The aggregate bandwidth is the sum of both drives — a 9 GB/s + 3 GB/s pair reads experts ~33% faster than the fast drive alone, and the OMP-parallel pin/warmup load streams from both. Details worth knowing:
|
||||
|
||||
- the mirror is **validated at startup** (per-file size + safetensors header must be byte-identical to the primary); divergent or missing files silently stay on the primary, so a **partial mirror is fine** — a smaller second SSD holding only some shards still helps;
|
||||
- the mirror is **never written**: `.coli_usage`, `.coli_kv` and all sidecars stay on the primary;
|
||||
- a read error on the mirror falls back to the primary (one warning, no crash), so unplugging the second drive mid-run degrades instead of killing the server;
|
||||
- routing never changes tokens — both copies are byte-identical, and the per-run `MIRROR:` stats line shows GB served per drive.
|
||||
|
||||
The same engine spans the whole range: on a 25 GB laptop everything streams from
|
||||
disk (slow but correct); on a large host the entire expert set becomes resident
|
||||
(`CUDA_EXPERT_GB=auto PIN_GB=all`) and disk drops out of the decode path
|
||||
@@ -110,6 +148,13 @@ on-device across layers so the CPU expert loop runs uninterrupted; on Apple
|
||||
Silicon an experimental [Metal backend](docs/metal.md) does the batched expert
|
||||
math on the unified-memory GPU.
|
||||
|
||||
> **On real NVMe, measure `DIRECT=1`.** O_DIRECT bypasses the page cache and is
|
||||
> often a large win on drives with DRAM cache and bandwidth headroom (+34%
|
||||
> decode measured with `PIPE=1` on a Blackwell/Windows box; 4.25→9.69 GB/s in
|
||||
> iobench on a GB10) — but it is drive-dependent: QLC/DRAM-less or virtualised
|
||||
> disks can be neutral to negative. Try it first; keep what your hardware
|
||||
> rewards.
|
||||
|
||||
### Faithful model, compressed state
|
||||
|
||||
The forward pass is validated **token-exact against a `transformers` oracle**
|
||||
@@ -157,14 +202,9 @@ scale-granularity/rotation ablations live in
|
||||
|
||||
## Get started
|
||||
|
||||
> **New here, or on Windows?** The [Quick Start guide](docs/quickstart.md) walks
|
||||
> through install → build → model → first chat step by step for Linux, Windows,
|
||||
> and macOS. On **Windows** you don't even need to build: download the
|
||||
> `colibri-<version>-windows-x86_64.zip` from
|
||||
> [Releases](https://github.com/JustVugg/colibri/releases), unzip it, rename
|
||||
> `colibri-*-windows-x86_64.exe` → `glm.exe`, install
|
||||
> [Python 3](https://www.python.org/downloads/), and run `coli chat` — full
|
||||
> details in the [Windows section](docs/quickstart.md#windows).
|
||||
> **New here?** The [Quick Start guide](docs/quickstart.md) walks through
|
||||
> install → build → model → first chat step by step for Linux, Windows, and
|
||||
> macOS, with copy-paste commands and no assumed background.
|
||||
|
||||
### 1. Get the model
|
||||
|
||||
@@ -198,6 +238,17 @@ COLI_MODEL=/nvme/glm52_i4 ./coli doctor # read-only readiness check
|
||||
The engine at runtime is pure C — python is only used by the one-time converter
|
||||
and the optional API gateway.
|
||||
|
||||
**On Windows?** You don't need to build. Download the
|
||||
`colibri-<version>-windows-x86_64.zip` from
|
||||
[Releases](https://github.com/JustVugg/colibri/releases), unzip it, rename
|
||||
`colibri-*-windows-x86_64.exe` → `glm.exe` (so the `coli` launcher finds the
|
||||
engine), install [Python 3](https://www.python.org/downloads/), then run
|
||||
`coli chat`. Full walkthrough in the [Quick Start guide](docs/quickstart.md#windows).
|
||||
|
||||
Prefer a `coli` command on your PATH? From a checkout, `pip install -e .`
|
||||
registers it (the engine itself still lives in `c/` — this is an editable
|
||||
install from the clone, not a standalone wheel).
|
||||
|
||||
### 3. Go deeper
|
||||
|
||||
| topic | doc |
|
||||
@@ -211,6 +262,18 @@ and the optional API gateway.
|
||||
| Grammar-forced drafts (structured output) | [docs/grammar-draft.md](docs/grammar-draft.md) |
|
||||
| Environment variable inventory | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
|
||||
|
||||
## What's next
|
||||
|
||||
- **Algorithmic research is active.** The current hierarchy is LRU + a learned
|
||||
pin set; the next step is under way — smarter placement and scheduling,
|
||||
overlap of CPU and GPU expert execution, and routing-aware speculation.
|
||||
Everything lands the way this project always works: measured, reviewed, and
|
||||
merged in the open.
|
||||
- **More open models.** The tiering algorithm is model-agnostic: any MoE with
|
||||
routed experts can be staged the same way. GLM-5.2 and OLMoE run today;
|
||||
support for more open-weight families — **Kimi K2** (Moonshot AI),
|
||||
**Qwen3 MoE** (Alibaba), **MiniMax** — is on the roadmap.
|
||||
|
||||
## Supporting the project
|
||||
|
||||
colibrì started as a one-person project on a 12-core laptop with 25 GB of RAM;
|
||||
@@ -251,6 +314,14 @@ The hummingbird weighs a few grams, hovers in place, and visits a thousand
|
||||
flowers a day. This engine keeps a 744-billion-parameter giant alive on
|
||||
hummingbird rations: 25 GB of RAM, twelve CPU cores, and a lot of disk patience.
|
||||
|
||||
## Acknowledgements
|
||||
|
||||
colibrì is an engine; the minds it runs are a gift. Thank you to the teams
|
||||
releasing frontier-class weights in the open — **Z.ai** (GLM), **Moonshot AI**
|
||||
(Kimi), **Alibaba Qwen**, **MiniMax**, and **Allen AI** (OLMoE) — and to every
|
||||
contributor who benchmarked, bisected, replicated an atlas run, or sent a patch.
|
||||
This project is proof of what open weights make possible.
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0. GLM-5.2 weights are released by Z.ai under MIT.
|
||||
|
||||
+237
@@ -0,0 +1,237 @@
|
||||
<p align="center">
|
||||
<img src="assets/colibri.svg" width="500" alt="colibrì——小巧引擎,庞大模型">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · 简体中文 · <a href="README.zh-TW.md">繁體中文</a> · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**小巧引擎,庞大模型。**只需约 25 GB 内存,就能在消费级电脑上运行 **GLM-5.2(744B 参数的 MoE)**——以零依赖的纯 C 实现,从磁盘流式加载专家。
|
||||
|
||||
Colibrì 是一套轻量、保持模型质量的 MoE 运行时,将 VRAM、RAM
|
||||
与存储设备视为统一管理的内存层级。高速内存不足可能降低速度,
|
||||
但默认策略**绝不会在未告知的情况下改变模型精度或路由语义**。
|
||||
|
||||
```
|
||||
$ ./coli chat
|
||||
🐦 colibrì v1.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
|
||||
✓ ready in 32s · resident 9.9 GB
|
||||
› ciao!
|
||||
◆ Ciao! 😊 Come posso aiutarti oggi?
|
||||
```
|
||||
|
||||
## 实际运行效果
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-dashboard.png" width="900" alt="colibrì 网页仪表盘——实时指标、硬件面板与专家存储层级">
|
||||
</p>
|
||||
<p align="center"><em>网页仪表盘(<code>./coli web</code>):744B 模型达到 <strong>4 tok/s、TTFT 1.6 秒、磁盘读取 0</strong>——
|
||||
在 6× RTX 5090 上让所有专家常驻,并实时显示 token 指标、每轮耗时明细、
|
||||
VRAM/RAM/磁盘层级条,以及角落的实时迷你大脑。</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-brain.png" width="900" alt="大脑页面——以实时皮层呈现 19,456 个专家">
|
||||
</p>
|
||||
<p align="center"><em><strong>大脑(Brain)</strong>页面:将全部 19,456 个专家呈现为活的皮层——颜色代表存储层级,
|
||||
亮度代表路由热度,每轮被路由到的专家都会闪白。将光标停在专家上,即可查看其
|
||||
<a href="https://github.com/JustVugg/colibri/issues/175">实测主题亲和度</a>。</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-atlas.png" width="900" alt="图谱页面——以 3D 星系呈现实测专家图谱">
|
||||
</p>
|
||||
<p align="center"><em><strong>图谱(Atlas)</strong>页面:将<a href="https://github.com/JustVugg/colibri/issues/175">实测专家图谱</a>
|
||||
呈现为 3D 星系——共 13,260 个已分析专家,其中 1,041 个可复现的专门专家会按主题聚集
|
||||
(诗歌、法律、中文、SQL……)。位置取自实测路由亲和度,而非学习出的嵌入向量。拖拽即可旋转。</em></p>
|
||||
|
||||
## 核心概念
|
||||
|
||||
744B 的专家混合(Mixture-of-Experts)模型,每个 token 只会激活约 40B 参数——
|
||||
其中每个 token 之间会变动的只有约 11 GB(被路由到的专家):
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/sparse.png" width="880" alt="每个 token 只会激活约 5.4% 的参数">
|
||||
</p>
|
||||
|
||||
所以模型不必完整**装进**高速内存,而是需要正确**放置**:
|
||||
|
||||
- **稠密部分**(注意力、共享专家、嵌入——约 17B 参数)以 int4
|
||||
**常驻 RAM**(约 9.9 GB);
|
||||
- **19,456 个路由专家**(75 个 MoE 层 × 256,加上 MTP head;每个在 int4 下约 19 MB)
|
||||
**存放在磁盘**(约 370 GB),并**按需流式加载**,配合逐层 LRU 缓存、
|
||||
会学习的热门专家固定存储区,以及可选的 VRAM 层级。
|
||||
|
||||
引擎是一个 C 主文件(`c/colibri.c`)加上若干头文件。不需要 BLAS,
|
||||
运行时不需要 Python,也不需要 GPU。
|
||||
|
||||
## 工作原理
|
||||
|
||||
### 每个 token 的处理路径
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/token-path.png" width="880" alt="路由 → 并集 → 放置 → 重叠执行 → 学习">
|
||||
</p>
|
||||
|
||||
每个 token 的每一层都会经过相同的五个步骤。设计目标是让
|
||||
**放置只决定速度**——无论专家是从 VRAM 还是磁盘响应,路由器的决策与权重精度都完全相同。
|
||||
|
||||
### 统一内存层级,取代单一内存门槛
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/tiers.png" width="880" alt="VRAM/RAM/NVMe 三层专家常驻架构">
|
||||
</p>
|
||||
|
||||
同一套引擎覆盖完整硬件范围:在 25 GB 笔记本上,一切都从磁盘流式加载
|
||||
(慢,但结果正确);在大内存主机上,则可让整组专家常驻
|
||||
(`CUDA_EXPERT_GB=auto PIN_GB=all`),让磁盘完全退出解码路径。
|
||||
两端之间有一层**学习型缓存**:引擎会记录*你的*工作负载路由到哪些专家
|
||||
(`.coli_usage`,每轮更新),并自动固定最热门的专家——colibrì 确实会越用越快。
|
||||
在多路主机上,`COLI_NUMA=1` 会将常驻权重交错分配到各内存控制器
|
||||
([#82](https://github.com/JustVugg/colibri/issues/82))。
|
||||
|
||||
### 绝不为同一次磁盘读取等待两遍
|
||||
|
||||
缓存未命中的代价很高,因此引擎大部分的巧思都用来避免或重叠这些读取:
|
||||
每个专家的三个矩阵相邻存储,并以一次 `pread` 读取;有界异步 I/O 池
|
||||
(`PIPE=1`,默认启用)会在常驻专家计算时加载缺失的专家;批量位置只读取每个
|
||||
不重复专家一次(**批量并集**);路由前瞻线程(`PILOT=1`)则预取下一层专家——
|
||||
实测显示,路由结果提前一层时有 **71.6% 的可预测性**。
|
||||
在 GPU 上,常驻管线(`COLI_CUDA_PIPE=2`)让残差流跨层保留在设备端,
|
||||
使 CPU 专家循环不中断;在 Apple Silicon 上,实验性的
|
||||
[Metal 后端](docs/metal.md)会用统一内存 GPU 执行批量专家运算。
|
||||
|
||||
### 忠实模型,压缩状态
|
||||
|
||||
前向传播已通过 `transformers` oracle 验证为**逐 token 完全一致**
|
||||
(teacher-forcing 32/32)。MLA 注意力存储压缩后的 KV 状态——每个 token 为 576 个
|
||||
浮点数,而非 32,768 个(**缩小 57×**)——并跨重启持久保存
|
||||
(`.coli_kv`):对话可暖启恢复,不需重新 prefill,结果与不中断的会话
|
||||
逐字节相同。DSA 稀疏注意力(GLM-5.2 的 lightning indexer)已忠实实现,
|
||||
并通过强制选取所有 key,验证可精确复现稠密注意力。
|
||||
|
||||
### 诚实的推测解码
|
||||
|
||||
GLM-5.2 原生 MTP head 会起草 token,再由主模型以一次批量前向传播验证——
|
||||
条件合适时每次 forward 可产生 2.2–2.8 个 token。两条来之不易的规则已成为默认值:
|
||||
MTP head 必须是 **int8**(int4 head 的接受率会崩塌到 0–4%,见
|
||||
[#8](https://github.com/JustVugg/colibri/issues/8)),且草稿与验证必须计算
|
||||
**相同函数**——`SPEC_PIN=1` 会把两者固定在同一 kernel family
|
||||
(完整取证过程见 [#163](https://github.com/JustVugg/colibri/issues/163))。
|
||||
语法强制草稿([`GRAMMAR=file.gbnf`](docs/grammar-draft.md))可在受限 JSON 输出中,
|
||||
以近乎免费的代价提高接受率。推测解码是否带来净收益取决于缓存热度——请实测,
|
||||
若不划算就使用 `DRAFT=0`。
|
||||
|
||||
## 实际成果
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/ladder.png" width="880" alt="各硬件级别的实测解码速度">
|
||||
</p>
|
||||
|
||||
同一套引擎、同一个 int4 容器——硬件只会改变专家的存放位置。
|
||||
[完整 benchmark 表格](docs/benchmarks.md)中的重点如下:
|
||||
|
||||
- **6× RTX 5090,全部常驻:**解码 5.8–6.8 tok/s,TTFT 约 13 秒
|
||||
([实验记录](docs/experiments/glm52-6x5090-2026-07-12.md));
|
||||
- **128 GB、仅使用 CPU 的台式机:**热缓存后约 1.8 tok/s
|
||||
([#200](https://github.com/JustVugg/colibri/issues/200));
|
||||
- **单张 RTX 5070 Ti 的笔记本级主机:**通过 GPU 常驻管线达到 1.07 tok/s
|
||||
([#273](https://github.com/JustVugg/colibri/issues/273));
|
||||
- **25 GB 开发机:**冷启动 0.05–0.1 tok/s——这是项目起步时已证实的下限,
|
||||
也仍是诚实的基准。
|
||||
|
||||
质量来自测量,而非假设:int4 容器的量化损失,以及 scale granularity/rotation
|
||||
消融实验,收录于 [docs/benchmarks.md](docs/benchmarks.md#quality-benchmark)、
|
||||
[#108](https://github.com/JustVugg/colibri/issues/108) 与
|
||||
[#81](https://github.com/JustVugg/colibri/issues/81)。
|
||||
|
||||
## 开始使用
|
||||
|
||||
### 1. 获取模型
|
||||
|
||||
Hugging Face 上已有预转换的 **GLM-5.2 int4** 容器——请务必使用
|
||||
**含 int8 MTP head 的版本**:
|
||||
|
||||
**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp**
|
||||
|
||||
> ⚠️ 原始镜像使用 int4 MTP head → 草稿接受率为 0%
|
||||
>([#8](https://github.com/JustVugg/colibri/issues/8))。请检查你的版本:
|
||||
> `ls -l <model>/out-mtp-*`——正确的 int8 大小为 `3527131672 / 5366238584 / 1065950496`。
|
||||
|
||||
你也可以自行从 FP8 源转换——只需一条可断点续传的命令,且任何时候都不需要
|
||||
在磁盘上同时存放完整的 756 GB:
|
||||
|
||||
```bash
|
||||
cd c && ./setup.sh # 检查 gcc/OpenMP、构建并运行自测
|
||||
./coli convert --model /nvme/glm52_i4 # 逐 shard 下载并转换(仅此一次需要 python)
|
||||
```
|
||||
|
||||
### 2. 运行
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat # 自动检测 RAM 预算、缓存与 MTP
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli plan # 查看规划的 VRAM/RAM/磁盘配置
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli doctor # 只读就绪检查
|
||||
./coli web --model /nvme/glm52_i4 # 在同一端口提供 API 与网页仪表盘
|
||||
./coli serve --model /nvme/glm52_i4 # 仅提供 OpenAI 兼容 API
|
||||
```
|
||||
|
||||
引擎运行时是纯 C——python 只供一次性转换工具与可选的 API gateway 使用。
|
||||
|
||||
### 3. 深入了解
|
||||
|
||||
| 主题 | 文档 |
|
||||
|---|---|
|
||||
| Benchmark、社区实测数据、质量测量 | [docs/benchmarks.md](docs/benchmarks.md) |
|
||||
| 调优选项、策略、学习型缓存、预取 | [docs/tuning.md](docs/tuning.md) |
|
||||
| Windows 11 原生构建(含 CUDA DLL) | [docs/windows.md](docs/windows.md) |
|
||||
| CUDA 后端、VRAM 专家层级、全部常驻 | [docs/cuda.md](docs/cuda.md) |
|
||||
| Apple Silicon Metal 后端 | [docs/metal.md](docs/metal.md) |
|
||||
| OpenAI 兼容 API、KV slots、网页仪表盘 | [docs/api.md](docs/api.md) |
|
||||
| 语法强制草稿(结构化输出) | [docs/grammar-draft.md](docs/grammar-draft.md) |
|
||||
| 环境变量完整清单 | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
|
||||
|
||||
## 支持项目
|
||||
|
||||
colibrì 最初由一人使用 12 核心、25 GB RAM 的笔记本开发;
|
||||
如今它的数据来自社区中各种真实机器。如果这个项目对你有用:
|
||||
|
||||
- ⭐ 为仓库加星并分享;
|
||||
- 🐛 以 issue 提交你的硬件 benchmark 数据——实测数据比任何其他事都更能推动项目;
|
||||
- 💬 若想赞助开发或捐赠硬件,请通过 GitHub issues 联系。
|
||||
|
||||
## 仓库结构
|
||||
|
||||
```
|
||||
Makefile 根目录构建/检查入口
|
||||
c/
|
||||
├── colibri.c 引擎主文件
|
||||
├── quant.h 量化 matmul 内核(SIMD 多架构)
|
||||
├── sample.h 采样与 stop-set 管理
|
||||
├── kv_persist.h .coli_kv 磁盘持久化
|
||||
├── telemetry.h 仪表盘协议、统计与用量持久化
|
||||
├── st.h, tok.h, json.h 运行时头文件
|
||||
├── backend_cuda.* 可选的 CUDA 层级
|
||||
├── Makefile 构建与本地检查
|
||||
├── coli 用户界面 CLI
|
||||
├── openai_server.py OpenAI 兼容 HTTP gateway
|
||||
├── setup.sh 一条命令完成本地设置
|
||||
├── tools/ 离线转换、fixtures 与 benchmarks
|
||||
├── scripts/ 长时间转换辅助工具
|
||||
└── tests/ 零依赖的 C 与 Python 测试
|
||||
web/ 浏览器 UI(纯 OpenAI API client)
|
||||
desktop/ 封装网页 UI 的 Tauri v2 桌面 shell
|
||||
docs/ 参考文档、实验与媒体文件
|
||||
```
|
||||
|
||||
运行时路径刻意保持扁平、易读:`colibri.c` 加上若干头文件。
|
||||
在仓库根目录执行 `make`、`make check` 与 `make clean`,
|
||||
都会转发给引擎的 Makefile。
|
||||
|
||||
## 为什么叫"colibrì"
|
||||
|
||||
蜂鸟只有几克重,能在原地悬停,并在一天内造访上千朵花。
|
||||
这套引擎只用蜂鸟般的配给,就能让 744B 参数的巨人运转:
|
||||
25 GB RAM、十二个 CPU 核心,以及对磁盘的大量耐心。
|
||||
|
||||
## 许可证
|
||||
|
||||
Apache 2.0。GLM-5.2 权重由 Z.ai 以 MIT 许可发布。
|
||||
+8
-4
@@ -3,7 +3,7 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · 繁體中文
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a> · 繁體中文 · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**小巧引擎,龐大模型。**只要約 25 GB 記憶體,就能在消費級電腦上執行 **GLM-5.2(744B 參數的 MoE)**——以零相依套件的純 C 實作,從硬碟串流載入專家。
|
||||
@@ -60,7 +60,7 @@ VRAM/RAM/硬碟層級長條,以及角落的即時迷你大腦。</em></p>
|
||||
**存放在硬碟**(約 370 GB),並**隨需串流載入**,搭配逐層 LRU 快取、
|
||||
會學習的熱門專家固定儲存區,以及選用的 VRAM 層級。
|
||||
|
||||
引擎由單一 C 檔(`c/glm.c`)與少量標頭檔組成。不需要 BLAS,
|
||||
引擎由主 C 檔(`c/colibri.c`)與多個標頭檔模組組成。不需要 BLAS,
|
||||
執行階段不需要 Python,也不需要 GPU。
|
||||
|
||||
## 運作方式
|
||||
@@ -203,7 +203,11 @@ colibrì 最初是由一人使用 12 核心、25 GB RAM 的筆電開發;
|
||||
```
|
||||
Makefile 根目錄建置/檢查入口
|
||||
c/
|
||||
├── glm.c 單檔 GLM 引擎
|
||||
├── colibri.c GLM 引擎主檔
|
||||
├── quant.h 量化 matmul kernel
|
||||
├── sample.h 取樣與 stop-set
|
||||
├── kv_persist.h .coli_kv 磁碟持久化
|
||||
├── telemetry.h 儀表板協定、統計
|
||||
├── st.h, tok.h, json.h 執行階段標頭檔
|
||||
├── backend_cuda.* 選用的 CUDA 層級
|
||||
├── Makefile 建置與本機檢查
|
||||
@@ -218,7 +222,7 @@ desktop/ 包裝網頁 UI 的 Tauri v2 桌面 shell
|
||||
docs/ 參考文件、實驗與媒體檔
|
||||
```
|
||||
|
||||
執行階段路徑刻意維持扁平、易讀:`glm.c` 加上少量標頭檔。
|
||||
執行階段路徑刻意維持扁平、易讀:`colibri.c` 加上模組化標頭檔。
|
||||
在儲存庫根目錄執行 `make`、`make check` 與 `make clean`,
|
||||
都會轉交給引擎的 Makefile。
|
||||
|
||||
|
||||
+356
@@ -0,0 +1,356 @@
|
||||
# E5 — MTLResidencySet over the existing malloc'd slabs (experiment branch)
|
||||
|
||||
Branch: `e5/metal-residency-set` (cut from `origin/dev` @ `caa49f7`, per spec — E4 was cut
|
||||
from `main` @ `72d3d37`; `backend_metal.mm`/`.h` are byte-identical between the two bases,
|
||||
confirmed via `git diff 72d3d37 origin/dev -- c/backend_metal.mm c/backend_metal.h`).
|
||||
|
||||
## The hypothesis
|
||||
|
||||
E4 (MTLHeap-backed slabs) proved that batching residency declaration kills the GPU stall
|
||||
(25.9s → 3.9s at cap16, −85%), but changing the *allocation* (heap sub-buffers instead of
|
||||
malloc'd host memory) brought a +12–13s expert-disk-load tax, suspected first-touch/lock
|
||||
contention on CPU-writes into GPU-owned heap pages. E5 decouples the two: keep the exact
|
||||
same malloc'd slabs and per-slab `newBufferWithBytesNoCopy`-wrapped `MTLBuffer`s, and change
|
||||
**only** residency bookkeeping — declare residency once, ahead of time, on a set attached to
|
||||
the command queue, instead of once per command buffer via `useResource:`. If the stall
|
||||
reduction survives without the load-path tax (malloc pages never change ownership), E5 wins.
|
||||
|
||||
## What changed
|
||||
|
||||
All mechanism code is confined to `c/backend_metal.mm` —
|
||||
`coli_metal_register`/`coli_metal_unregister`'s existing signatures and every call site in
|
||||
`colibri.c` (expert_load, uring_load_add, qalloc, kv_alloc, map_of_fd) are untouched; the
|
||||
residency-set bookkeeping lives entirely inside those two functions' existing bodies. The
|
||||
`colibri.c`/`backend_metal.h` touches are two, both coordinator-sanctioned: the validator
|
||||
round-1 instrumentation hook (`coli_metal_resset_stats` + the gate-on-only `METAL-RESSET:`
|
||||
stats line in `profile_print`) and the ported fslab-OOM unwind fix (see "Validator round 1
|
||||
fixes" item 4). Still a smaller diff shape than E4,
|
||||
which needed a new alloc/free API and four new `glm.c` call-site arms because it changed the
|
||||
allocation function itself.
|
||||
|
||||
Env-gated `COLI_METAL_RESSET=1`, default OFF, runtime `@available(macOS 15.0, *)` guard with
|
||||
a one-line stderr fallback when requested on an older OS or when residency-set creation
|
||||
fails. Gate off ⇒ every new branch is skipped and behavior is byte-for-byte the stock path
|
||||
(verified by inspection: `g_resset_enabled` starts `false` and nothing sets it except inside
|
||||
the `COLI_METAL_RESSET` `getenv` branch in `coli_metal_init`, so `resset_add`/`resset_remove`/
|
||||
`resset_flush` are no-ops and `moe_submit`'s `useResource:` loop runs unconditionally).
|
||||
|
||||
### Lifecycle (`c/backend_metal.mm`)
|
||||
|
||||
- **Init** (`coli_metal_init`, end of the existing pipeline-setup `@autoreleasepool`): if
|
||||
`COLI_METAL_RESSET=1` and `@available(macOS 15.0, *)`, create one
|
||||
`MTLResidencySetDescriptor` (`initialCapacity=4096`, a presize hint only), call
|
||||
`[g_dev newResidencySetWithDescriptor:desc error:&err]`, and `[g_queue addResidencySet:rs]`
|
||||
— one set, attached once, for the process lifetime. Failure (old OS or creation error)
|
||||
prints one stderr line and leaves `g_resset_enabled=false` — stock path.
|
||||
- **`coli_metal_register`**: after wrapping the buffer exactly as today
|
||||
(`newBufferWithBytesNoCopy`) and pushing the `g_slabs` entry under `g_slab_mtx` exactly as
|
||||
today, calls `resset_add(b)` **after dropping `g_slab_mtx`** but before returning.
|
||||
`resset_add` takes a dedicated `g_resset_mtx` (guarding only the set mutations and the
|
||||
dirty flag), calls `[rs addAllocation:b]` and sets `g_resset_dirty` — **it does not
|
||||
commit**. No Metal call ever runs under `g_slab_mtx` (validator round-1 fix; E4's audit
|
||||
round 2 identified mutex-over-live-Metal-call as the leading suspect for its +12s
|
||||
expert-disk regression). Re-registering a live base (no in-tree caller does today) drops
|
||||
the replaced wrapper from the set via `resset_remove(old)` before adding the new one
|
||||
(hazard-audit defensive fix — the set would otherwise retain the old buffer, and its
|
||||
pages' residency, forever), keeping set membership an exact mirror of `g_slabs`.
|
||||
- **`coli_metal_unregister`**: erases the `g_slabs` entry under `g_slab_mtx` (stashing the
|
||||
buffer), then calls `resset_remove(b)` **outside `g_slab_mtx`**, before returning.
|
||||
`resset_remove` (under `g_resset_mtx`) calls `[rs removeAllocation:b]` **and commits
|
||||
immediately** — no batching — because the caller frees the host memory right after the
|
||||
function returns. See UNCERTAINTIES for why this asymmetry is deliberate.
|
||||
- **`moe_submit`** (the one function whose `use` list — resolved expert weight/scale slabs —
|
||||
scales with LRU cache size): calls `resset_flush()` at the top (commits any pending adds
|
||||
from `resset_add`, under `g_resset_mtx` — it never touches the slab lock), then, if
|
||||
`g_resset_enabled`, **skips** the
|
||||
`for(auto&b:use) [e useResource:b usage:MTLResourceUsageRead];` loop entirely — residency
|
||||
is already guaranteed by the queue-attached set. Every other `useResource:` call site in the
|
||||
file (`bind_gemv`'s weight/scale buffers, `coli_metal_attn_decode`/`coli_metal_layer_decode`'s
|
||||
`Lb`/`Rb`/`kvbW`/`kvbS`/`inB`/`pnB`/`rwB`/`rbB`, `coli_metal_gemm`'s `wb`/`sb`) is
|
||||
**left completely unchanged**, regardless of the flag — see "Why only `moe_submit`" below.
|
||||
- **Shutdown** (`coli_metal_shutdown`): `[g_queue removeResidencySet:rs]` then clears the
|
||||
globals, ahead of the existing `g_queue=nil; g_dev=nil;`.
|
||||
|
||||
### Why only `moe_submit` skips `useResource:`
|
||||
|
||||
Apple's `MTLResidencySet` class reference (developer.apple.com, fetched during design on
|
||||
2026-07-18) is explicit: *"Residency sets don't support hazard tracking, so you need to
|
||||
account for hazards with fences and events."* The SDK header on this box
|
||||
(`MTLResidencySet.h`, read directly) is **silent** on hazard tracking — the statement comes
|
||||
from Apple's online documentation and adoption guide ("Simplifying GPU resource management
|
||||
with residency sets"), not the header (see UNCERTAINTIES for sourcing). Dropping
|
||||
`useResource:` therefore risks losing whatever hazard-tracking value those calls provided. Rather than apply the residency set uniformly and
|
||||
argue *in general* that hazard tracking isn't load-bearing, this diff draws the line at the
|
||||
one call site the mechanism history actually implicates:
|
||||
|
||||
`moe_submit`'s `use` vector holds only **read-only** (`MTLResourceUsageRead`), **indirectly
|
||||
referenced** slab buffers — the kernel (`moe_gemv`) never touches them via `setBuffer:`; it
|
||||
dereferences raw GPU addresses (`waddr[e]`/`saddr[e]`) baked into a separately-bound address
|
||||
array (`bag`/`bau`/`bad`/`bsg`/`bsu`/`bsd`), which is exactly the "indirect reference" case
|
||||
`useResource:` exists for. No GPU-side write ever touches these buffers, so there is no
|
||||
write-after-write/read-after-write hazard for Metal's tracking to have been serializing in
|
||||
the first place; the one real hazard — a slab unregistered+freed+reused by the CPU while an
|
||||
async in-flight `moe_block_begin` command buffer still references it via a baked-in GPU
|
||||
address — is a **CPU-write race that Metal's hazard tracking never protected against anyway**
|
||||
(hazard tracking only covers GPU-side command dependencies visible through the Metal API; a
|
||||
raw host-memory write via `pread`/`memcpy` is invisible to it regardless of `useResource:`).
|
||||
That race is, and always was, the engine's own responsibility (slot/generation lifecycle: a
|
||||
slab isn't freed while an outstanding async handle still owns it) — unrelated to E5.
|
||||
|
||||
Every other call site (`bind_gemv`, attention K/V cache writes) either doesn't scale with
|
||||
cache size (fixed per-layer dense tensors — no perf benefit to touching) or has real
|
||||
GPU-side write traffic in the same encoder (`Lb`/`Rb` are written by `a_copy` and read by
|
||||
`a_score`/`a_clat` within one encoder — currently ordered by explicit
|
||||
`memoryBarrierWithScope:MTLBarrierScopeBuffers` calls already present in `encode_attention`,
|
||||
not by `useResource:`'s hazard tracking, but touching them wasn't needed for the hypothesis
|
||||
and was judged not worth the added surface area). Leaving them untouched keeps the diff's
|
||||
blast radius matched to the one seam the fix-plan's v5 finding actually names.
|
||||
|
||||
### Deferred-commit design (`resset_add` batches; `resset_remove` doesn't)
|
||||
|
||||
`coli_metal_register` is called from parallel OpenMP loader threads in tight bursts
|
||||
("warmup fan-out" — same phrase E4's audit used for the same threads). Committing on every
|
||||
single `addAllocation:` would reintroduce a per-slab cost on the load path, which is exactly
|
||||
what E4's own +12s regression looked like (mutex held across a live Metal call, serializing
|
||||
loader threads). So `resset_add` only marks `g_resset_dirty`; the commit is deferred to the
|
||||
next `moe_submit` call, which flushes once via `resset_flush()` before it relies on the set
|
||||
for residency.
|
||||
|
||||
This is correct — not just fast — because of an existing invariant the codebase already
|
||||
depends on for `resolve()` to work at all: a slab's `coli_metal_register` call always
|
||||
completes — including its trailing `resset_add`, which runs after `g_slab_mtx` is dropped
|
||||
but **before the function returns** — before any dispatch that references that slab's
|
||||
pointer can call `resolve()` for it (the caller in `colibri.c` cannot pass a freshly-loaded
|
||||
expert's pointer to a dispatch before the load — which registers it — returns). After the
|
||||
validator round-1 mutex split, the flush's synchronization runs through `g_resset_mtx`
|
||||
alone: `resset_add`'s set mutation + dirty write and `resset_flush`'s dirty read + commit
|
||||
are serialized by that one mutex, whose release/acquire pairs provide the memory ordering;
|
||||
`g_slab_mtx` still orders the slab-table bookkeeping (register-before-resolve) exactly as on
|
||||
stock. So any slab a given `moe_submit` invocation will resolve was `addAllocation:`-ed (and
|
||||
marked dirty) strictly before that invocation's `resset_flush()` acquired `g_resset_mtx` —
|
||||
the flush is guaranteed to cover it, regardless of what other threads are concurrently
|
||||
registering unrelated slabs. The two mutexes are never held simultaneously anywhere, so no
|
||||
deadlock ordering exists to maintain.
|
||||
|
||||
`resset_remove`, by contrast, commits synchronously and immediately, with no batching,
|
||||
because the caller (`colibri.c`, in every one of the four slab-realloc call sites, and in
|
||||
`kv_alloc`) frees the underlying host memory *right after* `coli_metal_unregister` returns.
|
||||
An uncommitted-but-still-set-member allocation pointing at memory the host has already freed
|
||||
is a potential use-after-free the GPU could act on — deferring that removal is not a
|
||||
performance-vs-safety tradeoff, it's just unsafe, so it isn't deferred. (The spec's own
|
||||
lifecycle wording backs this reading: "`coli_metal_register` → add allocation + commit
|
||||
**(batch commits where call pattern allows)**" carries a batching allowance that
|
||||
"`coli_metal_unregister` → remove + commit" does not.)
|
||||
|
||||
## Instrumentation parity
|
||||
|
||||
No existing counter's semantics changed. `coli_metal_moe_times`/`coli_metal_moe_counts`
|
||||
(`g_t_setup`, `g_t_gpu`, `g_t_kernel`, `g_t_scatter`, `g_moe_ok`/`g_moe_fb`/`g_moe_experts`)
|
||||
are computed exactly as before — `resset_flush()` runs *before* `ts_start = mnow()` in
|
||||
`moe_submit`, so its cost is **outside** `g_t_setup`, keeping the orchestrator's A/B harness
|
||||
reading the same counters with the same meaning across stock/E4/E5. The flush cost is
|
||||
surfaced separately (validator round-1 fix — the original design left it invisible, a blind
|
||||
spot for the battery): a dedicated `g_t_resset_flush` accumulator timed around the flush in
|
||||
`moe_submit`, exported via `coli_metal_resset_stats()` (`backend_metal.h`) and printed by
|
||||
`profile_print` as its own `METAL-RESSET: flush N.NNs` line — a **separate line following
|
||||
the `METAL:` line, mirroring E4's `METAL-HEAP:` convention, so the existing `METAL:` line
|
||||
the harness parses keeps its exact format** — printed **only when the gate is on** (the
|
||||
function returns 0 when off), so stock output stays byte-identical. The register-side
|
||||
`resset_add`/`resset_remove` costs have no dedicated counter: they run inside the engine's
|
||||
existing expert-load wait accounting (the `t_ewait` window in `colibri.c`), noted in a comment
|
||||
at `resset_add`, so a load-path regression from set bookkeeping would already show in the
|
||||
existing disk/wait numbers. `[METAL] residency-set: on` / the two fallback stderr lines from
|
||||
`coli_metal_init` confirm which path a run took.
|
||||
|
||||
## Validator round 1 fixes
|
||||
|
||||
1. **REQUIRED, Metal calls hoisted out of `g_slab_mtx`** (`backend_metal.mm`): the original
|
||||
design ran `addAllocation:`/`removeAllocation:`/`commit` while holding `g_slab_mtx`, the
|
||||
lock the parallel OMP loader threads contend on — structurally identical to the
|
||||
mutex-over-live-Metal-call shape E4's audit round 2 identified as the leading suspect for
|
||||
its replicated +12s expert-disk regression, and the SDK header notes commit on a resident
|
||||
set tries to make resources resident "instantly" (real synchronous work; this set is
|
||||
resident from startup since it is queue-attached for the process lifetime). Fixed by
|
||||
introducing a dedicated `g_resset_mtx` guarding only the set mutations + dirty flag;
|
||||
`g_slabs` push/erase stays under `g_slab_mtx` exactly as stock; the two mutexes are never
|
||||
held together. The register→flush→resolve happens-before argument is preserved — see the
|
||||
updated "Deferred-commit design" section and the comment at `resset_add`.
|
||||
2. **REQUIRED, false citations corrected** (this file + the `moe_submit` commit message,
|
||||
rewritten pre-push): the original text attributed the hazard-tracking and thread-safety
|
||||
statements to the SDK header (`MTLResidencySet.h`), which is in fact silent on both
|
||||
topics. The statements come from Apple's **online** `MTLResidencySet` class reference and
|
||||
the "Simplifying GPU resource management with residency sets" adoption guide (both
|
||||
fetched 2026-07-18 during design). All attributions now name the actual source; where a
|
||||
claim rests on design reasoning rather than documentation, it is labeled as such.
|
||||
3. **REQUIRED, flush cost made harness-visible**: `g_t_resset_flush` +
|
||||
`coli_metal_resset_stats()` + the gate-on-only `METAL-RESSET:` line in `profile_print` —
|
||||
see "Instrumentation parity" above.
|
||||
4. **Pre-existing fslab OOM-unwind bug — now CARRIED ON THIS BRANCH** (follow-up commit,
|
||||
coordinator-sanctioned second `colibri.c` change): `expert_load`'s fslab OOM path
|
||||
(`c/colibri.c`, in `expert_load_impl`) freed `s->slab` via `compat_aligned_free` **without**
|
||||
`coli_metal_unregister` — on stock that leaves a stale `g_slabs` entry whose GPU
|
||||
exposure ends with the last command buffer that declared it; under E5 the buffer would
|
||||
additionally be a **permanent residency-set member** referencing freed host memory until
|
||||
some later realloc of the same slot unregisters by pointer, a strictly longer-lived
|
||||
exposure than stock's transient per-CB one. Fixed by porting E4's reference
|
||||
implementation (`6753225`) to dev's non-heap code shape: `coli_metal_unregister(s->slab)`
|
||||
before the free. The `uring_load_add` analog (E4's audit round-2 "cheap insurance") is
|
||||
deliberately NOT carried: that arm is `#ifdef __linux__`-gated and `COLI_METAL` is
|
||||
macOS-only, so it is dead code on every real build target, and unlike E4 this branch has
|
||||
no allocation-path reason to touch the function at all.
|
||||
|
||||
## Per-seam differences vs E4
|
||||
|
||||
| Seam | E4 (`e4/metal-heap`) | E5 (this branch) |
|
||||
|---|---|---|
|
||||
| Allocation | New: `MTLHeap` sub-buffers via `coli_metal_heap_alloc` | Unchanged: same `posix_memalign` + `newBufferWithBytesNoCopy` |
|
||||
| Coordinator C source / `backend_metal.h` | `glm.c` touched (new alloc/free API, 4 call sites + `expert_host_release`) | `colibri.c` + header touched only for instrumentation and the OOM-unwind fix |
|
||||
| Residency scope | Declared once **per command buffer** (`useHeap:`, still inside `moe_submit`) | Declared once **for the process** (queue-attached set), refreshed incrementally at register/unregister |
|
||||
| Hazard tracking | Heap sub-buffers forced `MTLHazardTrackingModeUntracked` always (allocation-level) | Untouched at the resource level; `moe_submit` alone stops calling `useResource:` (encoder-level), independent of `COLI_METAL_UNTRACKED` |
|
||||
| Per-buffer vs per-set skip | `[b heap]` (Metal's own `MTLResource.heap` property) checked per buffer — heterogeneous mixes possible if a slab fell back to malloc | Blanket `if (!g_resset_enabled)` — homogeneous by construction, since every registered slab goes through the same `coli_metal_register` path when the gate is on |
|
||||
| Availability guard | None needed (`MTLHeap` is old API) | `@available(macOS 15.0, *)`, matching this box's macOS 26.5 but required for portability |
|
||||
| Known regression | +12–13s expert-disk load at cap16 (suspected first-touch/lock contention on heap pages) | None expected — malloc pages never change ownership; **unverified without a run** |
|
||||
|
||||
## What to measure (orchestrator, cap1/cap16, stock vs E4 vs E5)
|
||||
|
||||
1. **GPU stall** (`coli_metal_moe_times` gpu/kernel breakdown) — success: E5 ≈ E4's
|
||||
−85%-class reduction vs stock at cap16.
|
||||
2. **Expert-disk load path** (existing load/service-time counters) — success: E5 ≈ stock,
|
||||
i.e. **no** repeat of E4's +12–13s tax, since allocation is untouched.
|
||||
3. **tok/s** — should track (1) and (2) together.
|
||||
4. **md5 within a fixed dispatch composition** — flag on vs off must be byte-identical at a
|
||||
given cap (the "Output-invariant by construction" hard constraint); flag-on vs flag-on
|
||||
across cap1/cap16 may legitimately differ (different dispatch composition, per the
|
||||
fix-plan's "Determinism side-finding").
|
||||
5. **`[METAL] residency-set: on` line present in stderr** at flag-on startup, and absent
|
||||
(or the OS<15/create-failed fallback line) otherwise — cheap sanity check that a run
|
||||
actually exercised the intended path before trusting its numbers. Also read the
|
||||
**`METAL-RESSET: flush` line** (gate-on only): if that number is large, the deferred
|
||||
set-commit cost is eating the stall win from the dispatch side.
|
||||
6. If the hypothesis holds (E5 stall ≈ E4, E5 load-path ≈ stock, identical output), E5 becomes
|
||||
the upstream PR candidate and must include the cap-default recalibration flagged in PR
|
||||
#386's CURRENT-STATE CALIBRATION markers, per the spec's validation plan.
|
||||
|
||||
## Build
|
||||
|
||||
`cd c && make glm METAL=1` and a separate explicit `-Wall -Wextra` compile of
|
||||
`backend_metal.mm` (the Makefile's `METALXX` line does not itself pass `-Wall -Wextra`, so
|
||||
the warning surface was checked with those flags added explicitly; current `dev` contributes
|
||||
one pre-existing `unused variable 'TG'` warning), plus
|
||||
`cd c && make glm` (plain, non-Metal — the one `colibri.c` instrumentation touch, the `METAL-RESSET` stats line,
|
||||
is inside the pre-existing `#ifdef COLI_METAL` arm of `profile_print`, so the plain build
|
||||
compiles none of it), and
|
||||
`make metal-test` (existing synthetic kernel-correctness unit test — no model, no
|
||||
`glm52_i4/`, random weights — run once with `COLI_METAL_RESSET` unset and once with
|
||||
`COLI_METAL_RESSET=1` to numerically exercise `coli_metal_register`/`moe_submit`'s changed
|
||||
code path, since the task scope excludes running the real model). Exact results in the final
|
||||
report, not here (build results belong to the report per the task's deliverable split, and
|
||||
this file is written before the batched build run, per the scheduling constraint).
|
||||
|
||||
## UNCERTAINTIES
|
||||
|
||||
**Everything below is a judgment call, a seam where the residency-set lifecycle interacts
|
||||
with the existing queue/command-buffer structure, or something unverifiable without a real
|
||||
model run — flagged per the task's hard requirement.**
|
||||
|
||||
1. **The central design risk: skipping `useResource:` in `moe_submit` gives up Metal's
|
||||
automatic hazard tracking for that buffer set.** Sourcing (corrected in validator round
|
||||
1): the SDK header on this box
|
||||
(`/Library/Developer/CommandLineTools/SDKs/MacOSX.sdk/.../Headers/MTLResidencySet.h`,
|
||||
read directly) documents the protocol only in terms of residency and says nothing about
|
||||
hazard tracking either way; the two operative statements are from Apple's **online**
|
||||
documentation (fetched 2026-07-18): the "Simplifying GPU resource management with
|
||||
residency sets" adoption guide — *"You don't need to call `useResource`/`useHeap`... for
|
||||
allocations in a residency set"* — and the `MTLResidencySet` class reference —
|
||||
*"Residency sets don't support hazard tracking, so you need to account for hazards with
|
||||
fences and events."* I reasoned through every
|
||||
code path that touches `moe_submit`'s `use` buffers (read-only, indirectly referenced,
|
||||
never concurrently written, freed only after the engine's own slot lifecycle guarantees
|
||||
no outstanding async reference) and concluded removing `useResource:` there specifically
|
||||
is safe — but this reasoning is **not the same as having run the model**. If any code
|
||||
path I didn't trace lets a slab get unregistered while an async `moe_block_begin` handle
|
||||
is still in flight and reading it, this change removes a mitigation (weak as it may have
|
||||
been) that existed before. **This is the #1 thing to watch for md5 divergence on**, and
|
||||
the reason the scope was deliberately narrowed to `moe_submit` alone rather than applied
|
||||
uniformly.
|
||||
2. **Residency-set mutations are serialized under a dedicated `g_resset_mtx` (validator
|
||||
round-1 fix — originally they ran under `g_slab_mtx`, the E4-regression shape; no Metal
|
||||
call runs under the slab lock anymore).** The serialization itself is kept as required
|
||||
for correctness: Apple's online `MTLResidencySet` class reference states the set's
|
||||
*"methods aren't thread-safe"* (the SDK header contains no thread-safety statement either
|
||||
way — citation corrected in round 1; the online doc is the source). What remains
|
||||
**unverified without profiling a loaded run** is the *cost* of the calls themselves:
|
||||
`resset_remove`'s synchronous `commit` runs inside `coli_metal_unregister` on the
|
||||
loader path (its cost lands in the existing `t_ewait` accounting), and the SDK header
|
||||
says commit on a resident set tries to make added/removed resources resident/non-resident
|
||||
*"instantly"* — real synchronous work, since this set is resident from startup
|
||||
(queue-attached for the process lifetime). If `commit()`/`addAllocation:` turn out
|
||||
expensive on this hardware/OS build, the load path degrades through set bookkeeping
|
||||
rather than mutex contention — a different, now-decoupled failure mode, but the same
|
||||
symptom as E4's regression. Orchestrator: check E5's load-path timing against stock, not
|
||||
just against E4, and read the new `METAL-RESSET: flush` line for the dispatch-side share.
|
||||
3. **`resset_flush()`'s cost sits outside `g_t_setup`/the `moe_times` breakdown** (it runs
|
||||
before `ts_start = mnow()`), by design, to keep the harness's existing counters
|
||||
meaningful — and, since validator round 1, it is **no longer invisible**: the
|
||||
`g_t_resset_flush` accumulator surfaces it as the gate-on-only `METAL-RESSET: flush`
|
||||
line (see "Instrumentation parity"). Residual blind spots: (a) the accumulator is a
|
||||
plain double written from `moe_submit` on the engine thread, matching the existing
|
||||
`g_t_setup` convention — if `moe_submit` were ever called from multiple threads
|
||||
concurrently, both counters would be equally wrong; (b) the register-side
|
||||
`resset_add`/`resset_remove` costs have no dedicated counter and are only visible
|
||||
blended into the existing `t_ewait`/disk-wait numbers (comment at `resset_add` says so)
|
||||
— a fine-grained attribution would need a throwaway probe.
|
||||
4. **`initialCapacity = 4096` on the `MTLResidencySetDescriptor` is an unverified guess.**
|
||||
It's documented as a presize hint only (no correctness effect either way), chosen to be
|
||||
"clearly larger than the permanent-weight-tensor + KV-cache + plausible cap16 LRU-slab
|
||||
count" without actually counting those registrations precisely. Too small just means
|
||||
internal array growth; not a correctness concern, flagged only because it's a number I
|
||||
picked without measuring.
|
||||
5. **Not calling `requestResidency()` proactively.** Apple's guide frames it as an optional
|
||||
latency-hiding call ("call ahead of time during non-critical moments... to minimize [first
|
||||
command buffer] latency"), and Blender's Cycles PR (the spec's cited reference
|
||||
implementation) doesn't appear to use it either per its PR description. Omitted to keep
|
||||
the lifecycle minimal and match the reference pattern; if profiling shows a
|
||||
first-command-buffer-after-a-load-burst latency spike, this is the documented lever to try
|
||||
next, not implemented here.
|
||||
6. **The deferred-commit correctness argument (item in "Deferred-commit design" above) rests
|
||||
on a single-writer-before-single-reader program-order guarantee that is true today by
|
||||
inspection but is not an invariant enforced anywhere in code** (no assertion, no type-level
|
||||
guarantee) — it's the same kind of implicit ordering `resolve()` itself already depends on
|
||||
for correctness (a slab must be registered before any dispatch can resolve its pointer),
|
||||
so this diff doesn't introduce a new category of fragility, but it's worth naming
|
||||
explicitly rather than leaving implicit.
|
||||
7. **Async `moe_block_begin`/`moe_block_end` overlap with concurrent `register()` calls**
|
||||
(background loader threads registering new/different experts while an unrelated MoE block
|
||||
is still in flight on the GPU) was reasoned through but never exercised in a real
|
||||
concurrent stress scenario — the synthetic `metal-test` unit test's `run_moe` calls are
|
||||
single-threaded and synchronous (`coli_metal_moe_block`, not the async `_begin`/`_end`
|
||||
pair), so it does **not** cover this interleaving. The real engine's `PILOT`/prefetch and
|
||||
`moe_block_begin`/`_end` overlap path is exactly the concurrency shape most likely to
|
||||
expose a bug in this design if one exists, and is untested here by construction (out of
|
||||
scope: no model runs).
|
||||
8. **`coli_metal_gemm` (prefill path) and `bind_gemv` (attention path) still call
|
||||
`useResource:` unconditionally, so they get no CPU-overhead benefit from the residency set
|
||||
even though their buffers are also set members.** This is deliberate (see "Why only
|
||||
`moe_submit` skips" above) but means E5's win, if any, is scoped to the decode-path MoE
|
||||
dispatch loop specifically — prefill and attention timing should be unaffected by the flag,
|
||||
which is itself a testable prediction the orchestrator's harness can check.
|
||||
9. **API surface verified against this box's actual SDK headers**
|
||||
(`MTLResidencySet.h`, `MTLDevice.h`, `MTLCommandQueue.h`, `MTLAllocation.h`,
|
||||
`MTLResource.h` — all read directly, not from memory) and against Apple's own
|
||||
"Simplifying GPU resource management with residency sets" guide, so the method names/
|
||||
signatures (`newResidencySetWithDescriptor:error:`, `addResidencySet:`,
|
||||
`removeResidencySet:`, `addAllocation:`, `removeAllocation:`, `commit`) are
|
||||
high-confidence. What is **not** independently verified is runtime behavior beyond what
|
||||
the docs state and what the synthetic unit test exercises — no substitute for the
|
||||
orchestrator's real cap-sweep battery.
|
||||
10. **Pre-existing fslab OOM-unwind bug — carried on this branch** (follow-up commit; see
|
||||
"Validator round 1 fixes" item 4 for the full mechanism). The one-line
|
||||
unregister-before-free fix from E4's `6753225` is ported to dev's non-heap code shape,
|
||||
so the upstream PR built from E5 inherits it automatically. Residual notes: (a) the fix
|
||||
is only reachable through the fslab-OOM path (allocation failure mid-load), so it is
|
||||
untestable without an OOM-injection harness and cannot affect the orchestrator's
|
||||
controlled A/B runs at sane RAM headroom — carried as correctness insurance, verified by
|
||||
inspection + clean builds only; (b) the `__linux__`-gated `uring_load_add` analog is
|
||||
deliberately not carried (dead code on every real build target — rationale in the fixes
|
||||
section).
|
||||
@@ -3,3 +3,5 @@ glm_tiny/
|
||||
olmoe_hf/
|
||||
olmoe_i4/
|
||||
.build-config
|
||||
tests/test_st_mirror
|
||||
tests/test_st_pread
|
||||
|
||||
+78
-29
@@ -56,11 +56,16 @@ OMPL =
|
||||
endif
|
||||
CFLAGS = -O3 $(OMPC) -Wall -Wextra -Wno-unused-parameter -Wno-misleading-indentation -Wno-unused-function
|
||||
# Opt-in: ARCH=native appends -mcpu=native (arm64 clang uses -mcpu, not -march),
|
||||
# which unlocks the i8mm SMMLA int8/int4 dot kernels in glm.c. ARCH unset ->
|
||||
# which unlocks the i8mm SMMLA int8/int4 dot kernels in colibri.c. ARCH unset ->
|
||||
# no -mcpu, default build byte-identical. Apple clang knows apple-m4 / native.
|
||||
# For older X86_64 Macs (for example, Mac Pro 2019) we need to use -march
|
||||
ifneq ($(ARCH),)
|
||||
ifneq (,$(X86_64))
|
||||
CFLAGS += -march=$(ARCH)
|
||||
else
|
||||
CFLAGS += -mcpu=$(ARCH)
|
||||
endif
|
||||
endif
|
||||
LDFLAGS = -lm $(OMPL)
|
||||
EXE =
|
||||
else ifneq ($(IS_WIN),)
|
||||
@@ -70,7 +75,7 @@ else ifneq ($(IS_WIN),)
|
||||
# ARCH default = x86-64-v3 (portable binary with AVX2). For max speed on THIS
|
||||
# machine use ARCH=native: on AVX-VNNI CPUs (Intel Alder Lake+, Meteor Lake+)
|
||||
# it also unlocks the 128-bit VPDPBUSD int8/int4 dot kernel (dot_i8i8/dot_i4i8),
|
||||
# which the x86-64-v3 baseline does not define. The #ifdef guards in glm.c mean
|
||||
# which the x86-64-v3 baseline does not define. The #ifdef guards in colibri.c mean
|
||||
# a v3 build simply compiles out the VNNI path - safe on any x86-64.
|
||||
CC = gcc
|
||||
ARCH ?= x86-64-v3
|
||||
@@ -166,7 +171,7 @@ else
|
||||
PYTHON ?= python3
|
||||
endif
|
||||
CUDA_OBJ =
|
||||
TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_st_pread$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_grouped$(EXE) tests/test_stops$(EXE) tests/test_topp$(EXE) tests/test_sample_nan$(EXE) tests/test_kv_alloc$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) tests/test_dsa_select$(EXE) tests/test_logit_nan$(EXE)
|
||||
TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_st_mirror$(EXE) tests/test_st_pread$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_grouped$(EXE) tests/test_stops$(EXE) tests/test_topp$(EXE) tests/test_sample_nan$(EXE) tests/test_tok_o200k$(EXE) tests/test_kv_alloc$(EXE) tests/test_int3$(EXE) tests/test_int3_load$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) tests/test_dsa_select$(EXE) tests/test_logit_nan$(EXE) tests/test_pipe_block$(EXE)
|
||||
ifneq (,$(LINUX))
|
||||
TEST_BINS += tests/test_uring$(EXE)
|
||||
endif
|
||||
@@ -207,17 +212,18 @@ LDFLAGS += -framework Metal -framework Foundation -lc++
|
||||
METAL_OBJ = backend_metal.o
|
||||
endif
|
||||
|
||||
all: glm$(EXE)
|
||||
all: colibri$(EXE)
|
||||
|
||||
# phony 'glm' → 'glm.exe' on Windows (so 'make glm' and 'coli build' work on every platform)
|
||||
glm: glm$(EXE)
|
||||
# phony targets — 'glm' kept for backward compatibility
|
||||
colibri: colibri$(EXE)
|
||||
glm: colibri$(EXE)
|
||||
|
||||
# Config stamp: make only tracks file timestamps, not flag changes. Without this,
|
||||
# `make glm.exe CUDA_DLL=1` after a prior CPU-only build reports "up to date" and
|
||||
# silently keeps the CPU-only binary (no CUDA loader) — a build that looks like it
|
||||
# worked but isn't. We record the build-affecting flags in .build-config and rewrite
|
||||
# it ONLY when they change (evaluated here at parse time, so the file's timestamp
|
||||
# moves exactly when the config moves). glm.exe and the CUDA/loader objects depend
|
||||
# `make colibri.exe CUDA_DLL=1` after a prior CPU-only build reports "up to date"
|
||||
# and silently keeps the CPU-only binary (no CUDA loader) — a build that looks like
|
||||
# it worked but isn't. We record the build-affecting flags in .build-config and
|
||||
# rewrite it ONLY when they change (evaluated here at parse time, so the file's
|
||||
# timestamp moves exactly when the config moves). The binary and CUDA/loader objects depend
|
||||
# on it, so they relink on a config change and stay put otherwise. (#306)
|
||||
BUILD_CONFIG := $(CC)|$(CFLAGS)|$(LDFLAGS)|CUDA=$(CUDA)|CUDA_DLL=$(CUDA_DLL)|ARCH=$(ARCH)|CUDA_ARCH=$(CUDA_ARCH)|METAL=$(METAL)
|
||||
BUILD_CONFIG_OLD := $(shell cat .build-config 2>/dev/null)
|
||||
@@ -226,8 +232,8 @@ $(shell printf '%s' '$(BUILD_CONFIG)' > .build-config)
|
||||
endif
|
||||
.build-config: ;
|
||||
|
||||
glm$(EXE): glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h $(CUDA_OBJ) $(METAL_OBJ) .build-config
|
||||
$(CC) $(CFLAGS) glm.c $(CUDA_OBJ) $(METAL_OBJ) -o glm$(EXE) $(LDFLAGS)
|
||||
colibri$(EXE): colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h sample.h kv_persist.h telemetry.h $(CUDA_OBJ) $(METAL_OBJ) .build-config
|
||||
$(CC) $(CFLAGS) colibri.c $(CUDA_OBJ) $(METAL_OBJ) -o colibri$(EXE) $(LDFLAGS)
|
||||
|
||||
# Windows runtime loader object: resolves coli_cuda_* from coli_cuda.dll.
|
||||
backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config
|
||||
@@ -272,8 +278,9 @@ olmoe$(EXE): olmoe.c st.h json.h compat.h
|
||||
# Use a baseline that matches the compiler target. macOS already targets a
|
||||
# portable baseline when ARCH is empty; forcing the x86 value there breaks
|
||||
# Apple Silicon. Unknown targets use native rather than an invalid x86 flag.
|
||||
# Intel Macs need -march for vector instructions
|
||||
ifneq (,$(DARWIN))
|
||||
PORTABLE_ARCH =
|
||||
PORTABLE_ARCH = $(if $(X86_64),x86-64-v3,)
|
||||
else ifneq (,$(AARCH64))
|
||||
PORTABLE_ARCH = armv8-a
|
||||
else ifneq (,$(PPC64))
|
||||
@@ -285,7 +292,7 @@ PORTABLE_ARCH = native
|
||||
endif
|
||||
|
||||
portable:
|
||||
$(MAKE) glm$(EXE) ARCH=$(PORTABLE_ARCH)
|
||||
$(MAKE) colibri$(EXE) ARCH=$(PORTABLE_ARCH)
|
||||
|
||||
iobench$(EXE): iobench.c compat.h
|
||||
$(CC) $(CFLAGS) iobench.c -o iobench$(EXE) $(LDFLAGS)
|
||||
@@ -293,12 +300,17 @@ iobench$(EXE): iobench.c compat.h
|
||||
tests/test_json$(EXE): tests/test_json.c json.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_tok_o200k$(EXE): tests/test_tok_o200k.c tok.h tok_unicode.h tok_unicode_o200k.h json.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
tests/test_st_pread$(EXE): tests/test_st_pread.c st.h json.h compat.h
|
||||
$(CC) $(CFLAGS) -DST_PREAD_CHUNK=7 $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_st$(EXE): tests/test_st.c st.h json.h compat.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_st_mirror$(EXE): tests/test_st_mirror.c st.h json.h compat.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_tier$(EXE): tests/test_tier.c tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
@@ -311,30 +323,36 @@ tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h j
|
||||
tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_idot$(EXE): tests/test_idot.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_idot$(EXE): tests/test_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_stops$(EXE): tests/test_stops.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_stops$(EXE): tests/test_stops.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_topp$(EXE): tests/test_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_topp$(EXE): tests/test_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
# bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test
|
||||
# gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp
|
||||
tests/bench_topp$(EXE): tests/bench_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/bench_topp$(EXE): tests/bench_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_sample_nan$(EXE): tests/test_sample_nan.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_logit_nan$(EXE): tests/test_logit_nan.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_int3$(EXE): tests/test_int3.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_int3_load$(EXE): tests/test_int3_load.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_logit_nan$(EXE): tests/test_logit_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
|
||||
@@ -343,15 +361,23 @@ tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
|
||||
tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_dsa_select$(EXE): tests/test_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_dsa_select$(EXE): tests/test_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
# bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356),
|
||||
# NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select
|
||||
tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_uring$(EXE): tests/test_uring.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
# bench_idot: microbenchmark (single-acc vs independent-acc AVX-VNNI idot), NOT a test gate.
|
||||
# Build on demand on an AVX-VNNI CPU: make tests/bench_idot ARCH=native
|
||||
tests/bench_idot$(EXE): tests/bench_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_uring$(EXE): tests/test_uring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_pipe_block$(EXE): tests/test_pipe_block.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
test-c: $(TEST_BINS)
|
||||
@@ -362,18 +388,41 @@ test-python:
|
||||
|
||||
test: test-c test-python
|
||||
|
||||
# --- Efficiency / regression suite (issue: "test the program for inefficiencies") ---
|
||||
# The tiny-model assertions live in test_inefficiency.py and run as part of
|
||||
# test-python (they're discovered by the test_*.py glob). These targets are
|
||||
# convenience entry points; the opt-in full-model report is NEVER in `make test`.
|
||||
#
|
||||
# make efficiency tiny-model asserted regression tests (CPU; part of test-python)
|
||||
# make efficiency-cuda the CUDA-path tests (requires a CUDA build — see below)
|
||||
# make efficiency-report opt-in full-model 🟢/🔴 diagnostic, never fails CI
|
||||
#
|
||||
# CUDA build (Windows): the CUDA tests need a host built with -DCOLI_CUDA plus
|
||||
# the runtime DLL. Do this FIRST — note CUDA_DLL=1 on BOTH the host and the
|
||||
# rule below, or `make glm.exe` will rebuild a CPU-only host and overwrite it:
|
||||
# make clean && make glm.exe CUDA_DLL=1 && make cuda-dll
|
||||
# The tests auto-skip with a clear message if the host is CPU-only.
|
||||
efficiency: test-python
|
||||
$(PYTHON) -m unittest tests.test_inefficiency -v
|
||||
|
||||
efficiency-cuda:
|
||||
$(PYTHON) -m unittest tests.test_inefficiency.TinyCudaEfficiencyTest -v
|
||||
|
||||
efficiency-report:
|
||||
$(PYTHON) tests/test_efficiency_report.py
|
||||
|
||||
# Local validation: one portable CPU build and dependency-free tests.
|
||||
check:
|
||||
$(MAKE) clean
|
||||
$(MAKE) portable
|
||||
$(MAKE) test
|
||||
|
||||
install: glm$(EXE) olmoe$(EXE)
|
||||
install: colibri$(EXE) olmoe$(EXE)
|
||||
$(INSTALL) -d $(DESTDIR)$(BINDIR)
|
||||
$(INSTALL) -d $(DESTDIR)$(LIBEXECDIR)
|
||||
$(INSTALL) -d $(DESTDIR)$(LIBEXECDIR)/tools
|
||||
$(INSTALL) -m 755 coli $(DESTDIR)$(BINDIR)/coli
|
||||
$(INSTALL) -m 755 glm$(EXE) $(DESTDIR)$(LIBEXECDIR)/glm$(EXE)
|
||||
$(INSTALL) -m 755 colibri$(EXE) $(DESTDIR)$(LIBEXECDIR)/colibri$(EXE)
|
||||
$(INSTALL) -m 755 olmoe$(EXE) $(DESTDIR)$(LIBEXECDIR)/olmoe$(EXE)
|
||||
$(INSTALL) -m 644 resource_plan.py doctor.py openai_server.py $(DESTDIR)$(LIBEXECDIR)/
|
||||
$(INSTALL) -m 644 tools/*.py $(DESTDIR)$(LIBEXECDIR)/tools/
|
||||
@@ -387,4 +436,4 @@ clean:
|
||||
|
||||
bench: iobench$(EXE)
|
||||
@if [ -n "$(ARGS)" ]; then ./iobench$(EXE) $(ARGS); else echo "built iobench$(EXE) — run: ./iobench$(EXE) <file> <MB> <iters> <threads> <direct 0|1>"; fi
|
||||
.PHONY: all glm cuda-test cuda-bench cuda-dll portable test-c test-python test check clean install uninstall bench
|
||||
.PHONY: all colibri glm cuda-test cuda-bench cuda-dll portable test-c test-python test check clean install uninstall bench
|
||||
|
||||
+352
-52
@@ -20,6 +20,9 @@ struct ColiCudaTensor {
|
||||
float *scales;
|
||||
size_t weight_bytes;
|
||||
int fmt, I, O, device;
|
||||
int gs; /* quant group size; 0 = per-row scales (#334) */
|
||||
int ng; /* number of scale groups per row = ceil(I/gs) for fmt=4 */
|
||||
size_t scale_count; /* floats in `scales`: O per-row, O*ng grouped */
|
||||
int tracked;
|
||||
RaggedKVEntry ragged[512];
|
||||
int ragged_count;
|
||||
@@ -34,15 +37,17 @@ typedef struct {
|
||||
size_t qx_cap, qscale_cap;
|
||||
float *host_x,*host_y,*host_kv; size_t host_x_cap,host_y_cap,host_kv_cap;
|
||||
float *aq,*al,*ar,*ac; size_t aq_cap,al_cap,ar_cap,ac_cap;
|
||||
float *pipe_buf[24]; size_t pipe_cap[24]; /* scratch persistenti del resident pipeline */
|
||||
float *pipe_buf[27]; size_t pipe_cap[27]; /* scratch persistenti del resident pipeline */
|
||||
cudaStream_t stream;
|
||||
void *group_desc; size_t group_desc_cap;
|
||||
size_t tensor_count, tensor_bytes;
|
||||
int group_pending; size_t group_pending_bytes; /* async expert-group in flight (Inc.4) */
|
||||
} DeviceContext;
|
||||
|
||||
typedef struct {
|
||||
const void *g,*u,*d; const float *gs,*us,*ds;
|
||||
int gf,uf,df,rows,offset;
|
||||
int ggs,ugs,dgs; /* per-tensor quant group size; 0 = per-row scales (#334 fmt=4) */
|
||||
} GroupDesc;
|
||||
|
||||
static DeviceContext g_ctx[COLI_CUDA_MAX_DEVICES];
|
||||
@@ -54,6 +59,8 @@ static std::mutex g_group_stats_mu;
|
||||
static int cuda_ok(cudaError_t err, const char *what) {
|
||||
if (err == cudaSuccess) return 1;
|
||||
std::fprintf(stderr, "[CUDA] %s: %s\n", what, cudaGetErrorString(err));
|
||||
(void)cudaGetLastError(); /* consume the sticky error: a failed call must
|
||||
not poison the next launch's error check */
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -79,8 +86,9 @@ static int select_ctx(DeviceContext *ctx) {
|
||||
__host__ __device__ static size_t row_bytes(int fmt, int I) {
|
||||
if (fmt == 0) return (size_t)I * sizeof(float);
|
||||
if (fmt == 1) return (size_t)I;
|
||||
if (fmt == 2) return (size_t)(I + 1) / 2;
|
||||
if (fmt == 2 || fmt == 4) return (size_t)(I + 1) / 2; /* fmt=4: same packed int4 */
|
||||
if (fmt == 3) return (size_t)(I + 3) / 4;
|
||||
if (fmt == 4) return (size_t)(I + 1) / 2; /* grouped int4: nibbles like fmt 2 */
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -89,7 +97,7 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int
|
||||
if (fmt == 0) return reinterpret_cast<const float *>(base)[i];
|
||||
if (fmt == 1) return static_cast<float>(reinterpret_cast<const int8_t *>(base)[i]);
|
||||
const uint8_t *q = base;
|
||||
if (fmt == 2) {
|
||||
if (fmt == 2 || fmt == 4) { /* fmt=4: same nibble layout */
|
||||
uint8_t v = q[i >> 1];
|
||||
int n=(i&1)?(v>>4):(v&15); return static_cast<float>(n&8?n-16:n);
|
||||
}
|
||||
@@ -97,20 +105,45 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int
|
||||
return static_cast<float>(((v >> ((i & 3) * 2)) & 3) - 2);
|
||||
}
|
||||
|
||||
/* Scale for output `row`, input element `k`. fmt=4 (grouped int4) stores ng
|
||||
* scales per row at scales[row*ng + k/gs]; every other quantized format has
|
||||
* one scale per row at scales[row]. Mirrors quant_matmul's fmt==4 branch so the
|
||||
* attention absorb kernels apply per-group scales instead of the per-row
|
||||
* (fmt=2) semantic that crashed #298's g64 kv_b. */
|
||||
__device__ static float absorb_scale(const float *wscale, int fmt, int gs, int ng, int row, int k) {
|
||||
if (!fmt) return 1.f;
|
||||
if (fmt != 4) return wscale[row];
|
||||
int g = k / gs; if (g >= ng) g = ng - 1; /* tail of the last (partial) group */
|
||||
return wscale[(size_t)row * ng + g];
|
||||
}
|
||||
|
||||
__global__ static void offset_to_signed_s4(uint8_t *q,size_t n){
|
||||
size_t i=(size_t)blockIdx.x*blockDim.x+threadIdx.x;if(i<n)q[i]^=0x88;
|
||||
}
|
||||
|
||||
__global__ static void quant_matmul(float *y, const float *x, const void *weights,
|
||||
const float *scales, int fmt, int S, int I, int O,
|
||||
size_t rb) {
|
||||
size_t rb, int gs, int ng) {
|
||||
int o = blockIdx.x;
|
||||
int s = blockIdx.y;
|
||||
float sum = 0.0f;
|
||||
size_t row = (size_t)o * rb;
|
||||
const float *xs = x + (size_t)s * I;
|
||||
for (int i = threadIdx.x; i < I; i += blockDim.x)
|
||||
sum += xs[i] * weight_at(weights, fmt, row, i);
|
||||
if (fmt == 4) {
|
||||
/* Grouped int4: one f32 scale per gs elements along I (ng groups per row).
|
||||
* Scale layout: scales[o*ng + g]. Each thread strides through I, applying
|
||||
* the appropriate group scale as it crosses group boundaries. This matches
|
||||
* the CPU matmul_i4_grouped accumulation exactly. */
|
||||
const float *scl = scales + (size_t)o * ng;
|
||||
for (int i = threadIdx.x; i < I; i += blockDim.x) {
|
||||
int g = i / gs;
|
||||
if (g >= ng) g = ng - 1; /* tail elements in the last (partial) group */
|
||||
sum += xs[i] * weight_at(weights, fmt, row, i) * scl[g];
|
||||
}
|
||||
} else {
|
||||
for (int i = threadIdx.x; i < I; i += blockDim.x)
|
||||
sum += xs[i] * weight_at(weights, fmt, row, i);
|
||||
}
|
||||
|
||||
__shared__ float partial[256];
|
||||
partial[threadIdx.x] = sum;
|
||||
@@ -120,7 +153,7 @@ __global__ static void quant_matmul(float *y, const float *x, const void *weight
|
||||
__syncthreads();
|
||||
}
|
||||
if (!threadIdx.x)
|
||||
y[(size_t)s * O + o] = partial[0] * (fmt ? scales[o] : 1.0f);
|
||||
y[(size_t)s * O + o] = (fmt && fmt != 4) ? partial[0] * scales[o] : partial[0];
|
||||
}
|
||||
|
||||
__global__ static void silu_mul(float *gate, const float *up, size_t n) {
|
||||
@@ -296,13 +329,50 @@ __global__ static void grouped_down_w4(float *y,const float *x,const GroupDesc *
|
||||
if(!threadIdx.x)y[(size_t)(d.offset+s)*D+o]=p[0]*d.ds[o];
|
||||
}
|
||||
|
||||
/* fmt=4 grouped-int4 variants (#334): identical structure to the w4 kernels,
|
||||
* but the scale varies along the input dimension — sc[o*ng + i/gs], applied
|
||||
* per element inside the accumulation (gs is even, so a packed byte never
|
||||
* straddles a group). gs<=0 degrades to per-row (ng=1), so mixed fmt2/fmt4
|
||||
* groups run correctly through this one kernel family. */
|
||||
__global__ static void grouped_hidden_g4_dual(float *gate,float *up,const float *x,
|
||||
const GroupDesc *desc,int I,int D){
|
||||
int o=blockIdx.x,s=blockIdx.y,c=blockIdx.z;GroupDesc d=desc[c];if(s>=d.rows)return;
|
||||
const uint8_t *gr=(const uint8_t*)d.g+(size_t)o*((D+1)/2);
|
||||
const uint8_t *ur=(const uint8_t*)d.u+(size_t)o*((D+1)/2);
|
||||
int ggs=d.ggs>0?d.ggs:D, ugs=d.ugs>0?d.ugs:D;
|
||||
const float *gsc=d.gs+(size_t)o*(size_t)((D+ggs-1)/ggs);
|
||||
const float *usc=d.us+(size_t)o*(size_t)((D+ugs-1)/ugs);
|
||||
const float *xs=x+(size_t)(d.offset+s)*D;float ga=0,ua=0;
|
||||
for(int b=threadIdx.x;b<(D+1)/2;b+=blockDim.x){float g0,g1,u0,u1;unpack_s4(gr[b],&g0,&g1);unpack_s4(ur[b],&u0,&u1);
|
||||
int i=b*2;float gv=gsc[i/ggs],uv=usc[i/ugs];
|
||||
ga+=xs[i]*g0*gv;ua+=xs[i]*u0*uv;
|
||||
if(i+1<D){ga+=xs[i+1]*g1*gv;ua+=xs[i+1]*u1*uv;}}
|
||||
__shared__ float gp[256],upv[256];gp[threadIdx.x]=ga;upv[threadIdx.x]=ua;__syncthreads();
|
||||
for(int n=128;n;n>>=1){if(threadIdx.x<n){gp[threadIdx.x]+=gp[threadIdx.x+n];upv[threadIdx.x]+=upv[threadIdx.x+n];}__syncthreads();}
|
||||
if(!threadIdx.x){size_t z=(size_t)(d.offset+s)*I+o;gate[z]=gp[0];up[z]=upv[0];}
|
||||
}
|
||||
__global__ static void grouped_down_g4(float *y,const float *x,const GroupDesc *desc,int D,int I){
|
||||
int o=blockIdx.x,s=blockIdx.y,c=blockIdx.z;GroupDesc d=desc[c];if(s>=d.rows)return;
|
||||
const uint8_t *row=(const uint8_t*)d.d+(size_t)o*((I+1)/2);
|
||||
int dgs=d.dgs>0?d.dgs:I;
|
||||
const float *dsc=d.ds+(size_t)o*(size_t)((I+dgs-1)/dgs);
|
||||
const float *xs=x+(size_t)(d.offset+s)*I;float sum=0;
|
||||
for(int b=threadIdx.x;b<(I+1)/2;b+=blockDim.x){float a,z;unpack_s4(row[b],&a,&z);
|
||||
int i=b*2;float sv=dsc[i/dgs];
|
||||
sum+=xs[i]*a*sv;if(i+1<I)sum+=xs[i+1]*z*sv;}
|
||||
__shared__ float p[256];p[threadIdx.x]=sum;__syncthreads();
|
||||
for(int n=128;n;n>>=1){if(threadIdx.x<n)p[threadIdx.x]+=p[threadIdx.x+n];__syncthreads();}
|
||||
if(!threadIdx.x)y[(size_t)(d.offset+s)*D+o]=p[0];
|
||||
}
|
||||
|
||||
__global__ static void attention_absorb_kernel(float *ctx,const float *q,const float *latent,
|
||||
const float *rope,const void *weights,const float *wscale,
|
||||
int fmt,int H,int Q,int R,int V,int K,int T,float scale){
|
||||
int fmt,int H,int Q,int R,int V,int K,int T,float scale,
|
||||
int gs,int ng){
|
||||
int h=blockIdx.x,tid=threadIdx.x,rbase=h*(Q+V);extern __shared__ float sm[];
|
||||
float *qa=sm,*cl=qa+K,*scores=cl+K;
|
||||
for(int k=tid;k<K;k+=blockDim.x){float a=0;for(int d=0;d<Q;d++)
|
||||
a+=q[(size_t)h*(Q+R)+d]*weight_at(weights,fmt,(size_t)(rbase+d)*row_bytes(fmt,K),k)*(fmt?wscale[rbase+d]:1.f);qa[k]=a;}
|
||||
a+=q[(size_t)h*(Q+R)+d]*weight_at(weights,fmt,(size_t)(rbase+d)*row_bytes(fmt,K),k)*absorb_scale(wscale,fmt,gs,ng,rbase+d,k);qa[k]=a;}
|
||||
__syncthreads();
|
||||
for(int t=tid;t<T;t+=blockDim.x){float a=0;const float *lt=latent+(size_t)t*K,*rt=rope+(size_t)t*R;
|
||||
for(int k=0;k<K;k++)a+=qa[k]*lt[k];for(int d=0;d<R;d++)a+=q[(size_t)h*(Q+R)+Q+d]*rt[d];scores[t]=a*scale;}
|
||||
@@ -313,19 +383,20 @@ __global__ static void attention_absorb_kernel(float *ctx,const float *q,const f
|
||||
for(int k=tid;k<K;k+=blockDim.x){float a=0;for(int t=0;t<T;t++)a+=scores[t]*latent[(size_t)t*K+k];cl[k]=a;}
|
||||
__syncthreads();
|
||||
for(int v=tid;v<V;v+=blockDim.x){int row=rbase+Q+v;float a=0;size_t rb=row_bytes(fmt,K);
|
||||
for(int k=0;k<K;k++)a+=cl[k]*weight_at(weights,fmt,(size_t)row*rb,k);ctx[(size_t)h*V+v]=a*(fmt?wscale[row]:1.f);}
|
||||
for(int k=0;k<K;k++)a+=cl[k]*weight_at(weights,fmt,(size_t)row*rb,k)*absorb_scale(wscale,fmt,gs,ng,row,k);ctx[(size_t)h*V+v]=a;}
|
||||
}
|
||||
|
||||
__global__ static void attention_absorb_batch_kernel(float *ctx,const float *q,
|
||||
const float *latent,const float *rope,const void *weights,const float *wscale,
|
||||
int fmt,int S,int H,int Q,int R,int V,int K,int T,float scale){
|
||||
int fmt,int S,int H,int Q,int R,int V,int K,int T,float scale,
|
||||
int gs,int ng){
|
||||
int s=blockIdx.y,h=blockIdx.x,tid=threadIdx.x,nt=T-S+s+1,rbase=h*(Q+V);
|
||||
if(s>=S||nt<1)return;
|
||||
extern __shared__ float sm[];float *qa=sm,*cl=qa+K,*scores=cl+K,*red=scores+T;
|
||||
const float *qs=q+((size_t)s*H+h)*(Q+R);
|
||||
for(int k=tid;k<K;k+=blockDim.x){float a=0;for(int d=0;d<Q;d++)
|
||||
a+=qs[d]*weight_at(weights,fmt,(size_t)(rbase+d)*row_bytes(fmt,K),k)*
|
||||
(fmt?wscale[rbase+d]:1.f);qa[k]=a;}
|
||||
absorb_scale(wscale,fmt,gs,ng,rbase+d,k);qa[k]=a;}
|
||||
__syncthreads();
|
||||
for(int t=tid;t<nt;t+=blockDim.x){float a=0;const float *lt=latent+(size_t)t*K;
|
||||
const float *rt=rope+(size_t)t*R;for(int k=0;k<K;k++)a+=qa[k]*lt[k];
|
||||
@@ -343,8 +414,8 @@ __global__ static void attention_absorb_batch_kernel(float *ctx,const float *q,
|
||||
a+=scores[t]*latent[(size_t)t*K+k];cl[k]=a;}
|
||||
__syncthreads();
|
||||
for(int v=tid;v<V;v+=blockDim.x){int row=rbase+Q+v;float a=0;size_t rb=row_bytes(fmt,K);
|
||||
for(int k=0;k<K;k++)a+=cl[k]*weight_at(weights,fmt,(size_t)row*rb,k);
|
||||
ctx[((size_t)s*H+h)*V+v]=a*(fmt?wscale[row]:1.f);}
|
||||
for(int k=0;k<K;k++)a+=cl[k]*weight_at(weights,fmt,(size_t)row*rb,k)*absorb_scale(wscale,fmt,gs,ng,row,k);
|
||||
ctx[((size_t)s*H+h)*V+v]=a;}
|
||||
}
|
||||
|
||||
/* Independent device-resident KV sequence per row. lengths selects the valid
|
||||
@@ -455,7 +526,7 @@ extern "C" void coli_cuda_shutdown(void) {
|
||||
if (ctx->qx) cudaFree(ctx->qx);
|
||||
if (ctx->qscale) cudaFree(ctx->qscale);
|
||||
if(ctx->aq)cudaFree(ctx->aq);if(ctx->al)cudaFree(ctx->al);if(ctx->ar)cudaFree(ctx->ar);if(ctx->ac)cudaFree(ctx->ac);
|
||||
for(int b=0;b<24;b++) if(ctx->pipe_buf[b]) cudaFree(ctx->pipe_buf[b]);
|
||||
for(int b=0;b<27;b++) if(ctx->pipe_buf[b]) cudaFree(ctx->pipe_buf[b]);
|
||||
if (ctx->host_x) cudaFreeHost(ctx->host_x);
|
||||
if (ctx->host_y) cudaFreeHost(ctx->host_y);
|
||||
if (ctx->host_kv) cudaFreeHost(ctx->host_kv);
|
||||
@@ -503,40 +574,66 @@ extern "C" void coli_cuda_group_stats(uint64_t *calls, uint64_t *experts, uint64
|
||||
if(d2h_ms) *d2h_ms=g_group_d2h_ms;
|
||||
}
|
||||
|
||||
/* group size for the NEXT upload on this thread (fmt=4): routed through a
|
||||
* thread_local so the widely-wired upload signature (and the Windows DLL ABI)
|
||||
* stays untouched. pin_load uploads in parallel, hence thread_local. */
|
||||
static thread_local int g_upload_gs = 0;
|
||||
extern "C" int coli_cuda_tensor_upload_g(ColiCudaTensor **tensor,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int I, int O, int device, int gs);
|
||||
extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int I, int O, int device) {
|
||||
if (!tensor) return 0;
|
||||
if (*tensor) {
|
||||
/* Cached device copy: usable even when the caller's host pointers are
|
||||
* gone. CUDA_RELEASE_HOST slots null their host pointers after upload,
|
||||
* and with the old order (!weights checked first) every later matmul
|
||||
* on such a slot failed here — the GPU tier silently never computed
|
||||
* for host-released slab experts. */
|
||||
ColiCudaTensor *t = *tensor;
|
||||
int want_gs = (fmt==4 && g_upload_gs>0) ? g_upload_gs : 0;
|
||||
return t->fmt == fmt && t->I == I && t->O == O && t->device == device && t->gs == want_gs;
|
||||
}
|
||||
DeviceContext *ctx = find_ctx(device);
|
||||
if (!tensor || !weights || I < 1 || O < 1 || !select_ctx(ctx)) return 0;
|
||||
if (!weights || I < 1 || O < 1 || !select_ctx(ctx)) return 0;
|
||||
size_t rb = row_bytes(fmt, I);
|
||||
if (!rb || (fmt && !scales)) return 0;
|
||||
if (*tensor) {
|
||||
ColiCudaTensor *t = *tensor;
|
||||
return t->fmt == fmt && t->I == I && t->O == O && t->device == device;
|
||||
}
|
||||
ColiCudaTensor *t = static_cast<ColiCudaTensor *>(std::calloc(1, sizeof(*t)));
|
||||
if (!t) return 0;
|
||||
t->fmt = fmt; t->I = I; t->O = O; t->device = device; t->weight_bytes = rb * (size_t)O;
|
||||
t->gs = (fmt==4 && g_upload_gs>0) ? g_upload_gs : 0;
|
||||
t->ng = t->gs ? (I + t->gs - 1) / t->gs : 1;
|
||||
t->scale_count = t->gs ? (size_t)O * (size_t)t->ng : (size_t)O;
|
||||
if (!cuda_ok(cudaMalloc(&t->weights, t->weight_bytes), "tensor allocation") ||
|
||||
!cuda_ok(cudaMemcpy(t->weights, weights, t->weight_bytes, cudaMemcpyHostToDevice), "tensor upload")) {
|
||||
coli_cuda_tensor_free(t);
|
||||
return 0;
|
||||
}
|
||||
if(fmt==2){offset_to_signed_s4<<<(unsigned)((t->weight_bytes+255)/256),256>>>((uint8_t*)t->weights,t->weight_bytes);
|
||||
if(fmt==2||fmt==4){ /* same nibble layout: offset-binary -> signed in place */
|
||||
offset_to_signed_s4<<<(unsigned)((t->weight_bytes+255)/256),256>>>((uint8_t*)t->weights,t->weight_bytes);
|
||||
if(!cuda_ok(cudaGetLastError(),"int4 weight conversion")){coli_cuda_tensor_free(t);return 0;}}
|
||||
if (fmt) {
|
||||
if (!cuda_ok(cudaMalloc(&t->scales, (size_t)O * sizeof(float)), "scale allocation") ||
|
||||
!cuda_ok(cudaMemcpy(t->scales, scales, (size_t)O * sizeof(float), cudaMemcpyHostToDevice), "scale upload")) {
|
||||
if (!cuda_ok(cudaMalloc(&t->scales, t->scale_count * sizeof(float)), "scale allocation") ||
|
||||
!cuda_ok(cudaMemcpy(t->scales, scales, t->scale_count * sizeof(float), cudaMemcpyHostToDevice), "scale upload")) {
|
||||
coli_cuda_tensor_free(t);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
t->tracked = 1;
|
||||
ctx->tensor_count++;
|
||||
ctx->tensor_bytes += t->weight_bytes + (fmt ? (size_t)O * sizeof(float) : 0);
|
||||
ctx->tensor_bytes += t->weight_bytes + (fmt ? t->scale_count * sizeof(float) : 0);
|
||||
*tensor = t;
|
||||
return 1;
|
||||
}
|
||||
extern "C" int coli_cuda_tensor_upload_g(ColiCudaTensor **tensor,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int I, int O, int device, int gs){
|
||||
g_upload_gs = gs>0 ? gs : 0;
|
||||
int r = coli_cuda_tensor_upload(tensor, weights, scales, fmt, I, O, device);
|
||||
g_upload_gs = 0;
|
||||
return r;
|
||||
}
|
||||
|
||||
extern "C" int coli_cuda_tensor_update(ColiCudaTensor *tensor,
|
||||
const void *weights,
|
||||
@@ -546,20 +643,35 @@ extern "C" int coli_cuda_tensor_update(ColiCudaTensor *tensor,
|
||||
if (!select_ctx(ctx)) return 0;
|
||||
if (!cuda_ok(cudaMemcpy(tensor->weights,weights,tensor->weight_bytes,
|
||||
cudaMemcpyHostToDevice),"tensor refresh")) return 0;
|
||||
if(tensor->fmt==2){
|
||||
if(tensor->fmt==2||tensor->fmt==4){
|
||||
offset_to_signed_s4<<<(unsigned)((tensor->weight_bytes+255)/256),256>>>(
|
||||
(uint8_t*)tensor->weights,tensor->weight_bytes);
|
||||
if(!cuda_ok(cudaGetLastError(),"int4 weight refresh")) return 0;
|
||||
}
|
||||
int ng = tensor->ng > 0 ? tensor->ng : 1;
|
||||
return !tensor->fmt || cuda_ok(cudaMemcpy(tensor->scales,scales,
|
||||
(size_t)tensor->O*sizeof(float),cudaMemcpyHostToDevice),"scale refresh");
|
||||
(tensor->scale_count?tensor->scale_count:(size_t)tensor->O)*sizeof(float),
|
||||
cudaMemcpyHostToDevice),"scale refresh");
|
||||
}
|
||||
|
||||
/* Test hook: COLI_GPU_FAIL_AFTER=N makes every GPU COMPUTE entry point report
|
||||
* failure after N successful calls (N=0: every call fails), exercising the
|
||||
* engine's CPU fallbacks and host-rematerialization end-to-end without real
|
||||
* hardware faults. Uploads/queries are not gated. Unset: no effect. */
|
||||
static long g_gpu_calls;
|
||||
static int fault_injected(void) {
|
||||
const char *fa = std::getenv("COLI_GPU_FAIL_AFTER");
|
||||
return fa && g_gpu_calls++ >= std::atol(fa);
|
||||
}
|
||||
|
||||
extern "C" int coli_cuda_matmul(ColiCudaTensor **tensor,
|
||||
float *y, const float *x,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int S, int I, int O, int device) {
|
||||
if (S < 1 || !coli_cuda_tensor_upload(tensor, weights, scales, fmt, I, O, device)) return 0;
|
||||
int fmt, int S, int I, int O, int device, int gs) {
|
||||
if (fault_injected()) return 0;
|
||||
if (S < 1) return 0;
|
||||
if (gs > 0) { if (!coli_cuda_tensor_upload_g(tensor, weights, scales, fmt, I, O, device, gs)) return 0; }
|
||||
else { if (!coli_cuda_tensor_upload(tensor, weights, scales, fmt, I, O, device)) return 0; }
|
||||
ColiCudaTensor *t = *tensor;
|
||||
DeviceContext *ctx = find_ctx(t->device);
|
||||
if (!select_ctx(ctx)) return 0;
|
||||
@@ -568,7 +680,7 @@ extern "C" int coli_cuda_matmul(ColiCudaTensor **tensor,
|
||||
if (!reserve(&ctx->x, &ctx->x_cap, xb) || !reserve(&ctx->y, &ctx->y_cap, yb)) return 0;
|
||||
if (!cuda_ok(cudaMemcpy(ctx->x, x, xb, cudaMemcpyHostToDevice), "input upload")) return 0;
|
||||
dim3 grid((unsigned)O, (unsigned)S);
|
||||
quant_matmul<<<grid, 256>>>(ctx->y, ctx->x, t->weights, t->scales, fmt, S, I, O, rb);
|
||||
quant_matmul<<<grid, 256>>>(ctx->y, ctx->x, t->weights, t->scales, fmt, S, I, O, rb, t->gs, t->ng);
|
||||
if (!cuda_ok(cudaGetLastError(), "matmul launch") ||
|
||||
!cuda_ok(cudaMemcpy(y, ctx->y, yb, cudaMemcpyDeviceToHost), "output download")) return 0;
|
||||
return 1;
|
||||
@@ -577,6 +689,7 @@ extern "C" int coli_cuda_matmul(ColiCudaTensor **tensor,
|
||||
extern "C" int coli_cuda_expert_mlp(ColiCudaTensor *gate, ColiCudaTensor *up,
|
||||
ColiCudaTensor *down, float *y,
|
||||
const float *x, int S) {
|
||||
if (fault_injected()) return 0;
|
||||
if (!gate || !up || !down || !x || !y || S < 1 ||
|
||||
gate->device != up->device || gate->device != down->device ||
|
||||
gate->I != up->I || gate->O != up->O ||
|
||||
@@ -591,13 +704,13 @@ extern "C" int coli_cuda_expert_mlp(ColiCudaTensor *gate, ColiCudaTensor *up,
|
||||
if (!cuda_ok(cudaMemcpy(ctx->x,x,xb,cudaMemcpyHostToDevice),"expert input upload")) return 0;
|
||||
dim3 hidden_grid((unsigned)I,(unsigned)S), output_grid((unsigned)D,(unsigned)S);
|
||||
quant_matmul<<<hidden_grid,256>>>(ctx->gate,ctx->x,gate->weights,gate->scales,
|
||||
gate->fmt,S,D,I,row_bytes(gate->fmt,D));
|
||||
gate->fmt,S,D,I,row_bytes(gate->fmt,D),gate->gs,gate->ng);
|
||||
quant_matmul<<<hidden_grid,256>>>(ctx->up,ctx->x,up->weights,up->scales,
|
||||
up->fmt,S,D,I,row_bytes(up->fmt,D));
|
||||
up->fmt,S,D,I,row_bytes(up->fmt,D),up->gs,up->ng);
|
||||
size_t n=(size_t)S*I;
|
||||
silu_mul<<<(unsigned)((n+255)/256),256>>>(ctx->gate,ctx->up,n);
|
||||
quant_matmul<<<output_grid,256>>>(ctx->y,ctx->gate,down->weights,down->scales,
|
||||
down->fmt,S,I,D,row_bytes(down->fmt,I));
|
||||
down->fmt,S,I,D,row_bytes(down->fmt,I),down->gs,down->ng);
|
||||
if (!cuda_ok(cudaGetLastError(),"expert MLP launch") ||
|
||||
!cuda_ok(cudaMemcpy(y,ctx->y,yb,cudaMemcpyDeviceToHost),"expert output download")) return 0;
|
||||
return 1;
|
||||
@@ -605,6 +718,7 @@ extern "C" int coli_cuda_expert_mlp(ColiCudaTensor *gate, ColiCudaTensor *up,
|
||||
|
||||
extern "C" int coli_cuda_shared_mlp_w4a16(ColiCudaTensor *gate,ColiCudaTensor *up,
|
||||
ColiCudaTensor *down,float *y,const float *x,int S){
|
||||
if (fault_injected()) return 0;
|
||||
if(!gate||!up||!down||!x||!y||S<1||gate->fmt!=2||up->fmt!=2||down->fmt!=2||
|
||||
gate->device!=up->device||gate->device!=down->device||gate->I!=up->I||
|
||||
gate->O!=up->O||down->I!=gate->O||down->O!=gate->I)return 0;
|
||||
@@ -636,19 +750,24 @@ extern "C" int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *downs,
|
||||
const int *rows, int count,
|
||||
float *y, const float *x) {
|
||||
if (fault_injected()) return 0;
|
||||
if (!gates || !ups || !downs || !rows || !x || !y || count < 1) return 0;
|
||||
ColiCudaTensor *first=gates[0];
|
||||
if (!first) return 0;
|
||||
int device=first->device,D=first->I,I=first->O,total=0,max_rows=0;
|
||||
GroupDesc host[64]; if(count>64) return 0;
|
||||
int all_s4=1;
|
||||
int all_s4=1,all_q4=1,any_g4=0;
|
||||
for(int c=0;c<count;c++){
|
||||
ColiCudaTensor *g=gates[c],*u=ups[c],*d=downs[c];
|
||||
if(!g||!u||!d||rows[c]<1||g->device!=device||u->device!=device||d->device!=device||
|
||||
g->I!=D||u->I!=D||g->O!=I||u->O!=I||d->I!=I||d->O!=D) return 0;
|
||||
host[c]={g->weights,u->weights,d->weights,g->scales,u->scales,d->scales,
|
||||
g->fmt,u->fmt,d->fmt,rows[c],total};
|
||||
g->fmt,u->fmt,d->fmt,rows[c],total,
|
||||
g->gs,u->gs,d->gs};
|
||||
all_s4&=g->fmt==2&&u->fmt==2&&d->fmt==2;
|
||||
all_q4&=(g->fmt==2||g->fmt==4)&&(u->fmt==2||u->fmt==4)&&(d->fmt==2||d->fmt==4)&&
|
||||
!(g->gs&1)&&!(u->gs&1)&&!(d->gs&1); /* even gs: a packed byte never straddles groups */
|
||||
any_g4|=g->fmt==4||u->fmt==4||d->fmt==4;
|
||||
total+=rows[c]; if(rows[c]>max_rows) max_rows=rows[c];
|
||||
}
|
||||
DeviceContext *ctx=find_ctx(device); if(!select_ctx(ctx)) return 0;
|
||||
@@ -689,7 +808,15 @@ extern "C" int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
quantize_s4_rows<<<total,256,0,ctx->stream>>>(ctx->qx,ctx->qscale,ctx->gate,total,I);
|
||||
grouped_s4_wmma<<<dim3((unsigned)((D+63)/64),(unsigned)count),256,0,ctx->stream>>>(ctx->y,ctx->qx,ctx->qscale,dev,I,D,2);
|
||||
}else if(all_s4&&ctx->compute_major>=7&&getenv("COLI_CUDA_TC_W4A16")&&
|
||||
atoi(getenv("COLI_CUDA_TC_W4A16"))){
|
||||
atoi(getenv("COLI_CUDA_TC_W4A16"))&&
|
||||
[&]{ int tc16_min=getenv("COLI_CUDA_TC_W4A16_MIN")?atoi(getenv("COLI_CUDA_TC_W4A16_MIN")):16;
|
||||
for(int c=0;c<count;c++) if(rows[c]>=tc16_min) return 1;
|
||||
return 0; }()){
|
||||
/* At least one expert has enough rows for a Tensor Core tile. Groups
|
||||
* where EVERY expert is below the threshold (decode: r=1) fall through
|
||||
* to the grouped-W4 path below — 3 launches for the whole group instead
|
||||
* of 4 per expert (#431: the launch flood measured at ~981 micro-kernels
|
||||
* per token came from decode riding this branch's per-expert fallback). */
|
||||
/* W4A16 Tensor Core per gruppo: attivazioni fp16 per tile (lossless al
|
||||
* contrario del path W4A4), un lancio per expert dentro lo stream —
|
||||
* l'overhead di lancio e' trascurabile rispetto ai GEMM. */
|
||||
@@ -711,12 +838,12 @@ extern "C" int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
/* piccoli batch: tile TC quasi vuoti + overhead di lancio — il
|
||||
* kernel naive per-elemento resta piu' veloce (misurato in decode) */
|
||||
quant_matmul<<<dim3((unsigned)I,(unsigned)r),256,0,ctx->stream>>>(g16,x16,
|
||||
host[c].g,host[c].gs,host[c].gf,r,D,I,row_bytes(host[c].gf,D));
|
||||
host[c].g,host[c].gs,host[c].gf,r,D,I,row_bytes(host[c].gf,D),0,1);
|
||||
quant_matmul<<<dim3((unsigned)I,(unsigned)r),256,0,ctx->stream>>>(u16,x16,
|
||||
host[c].u,host[c].us,host[c].uf,r,D,I,row_bytes(host[c].uf,D));
|
||||
host[c].u,host[c].us,host[c].uf,r,D,I,row_bytes(host[c].uf,D),0,1);
|
||||
silu_mul<<<(unsigned)(((size_t)r*I+255)/256),256,0,ctx->stream>>>(g16,u16,(size_t)r*I);
|
||||
quant_matmul<<<dim3((unsigned)D,(unsigned)r),256,0,ctx->stream>>>(y16,g16,
|
||||
host[c].d,host[c].ds,host[c].df,r,I,D,row_bytes(host[c].df,I));
|
||||
host[c].d,host[c].ds,host[c].df,r,I,D,row_bytes(host[c].df,I),0,1);
|
||||
}
|
||||
off16+=r;
|
||||
}
|
||||
@@ -730,7 +857,18 @@ extern "C" int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
}
|
||||
silu_mul<<<(unsigned)(((size_t)total*I+255)/256),256,0,ctx->stream>>>(ctx->gate,ctx->up,(size_t)total*I);
|
||||
grouped_down_w4<<<og,256,0,ctx->stream>>>(ctx->y,ctx->gate,dev,D,I);
|
||||
}else if(all_q4&&any_g4){
|
||||
/* grouped-int4 (fmt=4) present: per-group scales (#334). fmt=2 members
|
||||
* ride along as the ng=1 special case. */
|
||||
dim3 hg((unsigned)I,(unsigned)max_rows,(unsigned)count),og((unsigned)D,(unsigned)max_rows,(unsigned)count);
|
||||
grouped_hidden_g4_dual<<<hg,256,0,ctx->stream>>>(ctx->gate,ctx->up,ctx->x,dev,I,D);
|
||||
silu_mul<<<(unsigned)(((size_t)total*I+255)/256),256,0,ctx->stream>>>(ctx->gate,ctx->up,(size_t)total*I);
|
||||
grouped_down_g4<<<og,256,0,ctx->stream>>>(ctx->y,ctx->gate,dev,D,I);
|
||||
}else{
|
||||
/* generic path decodes fmt 0/1/2/3 only — a fmt=4 group that slipped the
|
||||
* gates above (odd gs) must NOT be silently decoded as int2 (#334). */
|
||||
for(int c=0;c<count;c++)
|
||||
if(host[c].gf==4||host[c].uf==4||host[c].df==4) return 0;
|
||||
dim3 hg((unsigned)I,(unsigned)max_rows,(unsigned)count),og((unsigned)D,(unsigned)max_rows,(unsigned)count);
|
||||
grouped_hidden<<<hg,256,0,ctx->stream>>>(ctx->gate,ctx->x,dev,I,D,0);
|
||||
grouped_hidden<<<hg,256,0,ctx->stream>>>(ctx->up,ctx->x,dev,I,D,1);
|
||||
@@ -757,10 +895,79 @@ extern "C" int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ---- Async expert group (Inc.4): issue/take split of coli_cuda_expert_group ----
|
||||
* The measured cost of the sync call at decode is ~0.45 ms/call of HOST-side wait
|
||||
* (stream sync + staging), vs ~0.18 ms of actual GPU work — 70% tax, paid ~5x per
|
||||
* layer because a token's 8 experts scatter across devices. issue() stages and
|
||||
* launches on the device stream and returns immediately; take() syncs and hands
|
||||
* back the pinned result rows. One issue may be outstanding per device; moe()
|
||||
* takes at each layer end, which also orders the next layer's reuse of the ctx
|
||||
* scratch buffers. Small batches only (decode/spec): bigger totals keep the sync
|
||||
* path with its TC variants. Numerics are the sync path's small-batch kernels,
|
||||
* so greedy output is byte-identical by construction. */
|
||||
extern "C" int coli_cuda_expert_group_issue(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs,
|
||||
const int *rows, int count,
|
||||
const float *x) {
|
||||
if (!gates || !ups || !downs || !rows || !x || count < 1 || count > 64) return 0;
|
||||
ColiCudaTensor *first=gates[0];
|
||||
if (!first) return 0;
|
||||
int device=first->device,D=first->I,I=first->O,total=0;
|
||||
GroupDesc host[64];
|
||||
for(int c=0;c<count;c++){
|
||||
ColiCudaTensor *g=gates[c],*u=ups[c],*d=downs[c];
|
||||
if(!g||!u||!d||rows[c]<1||g->device!=device||u->device!=device||d->device!=device||
|
||||
g->I!=D||u->I!=D||g->O!=I||u->O!=I||d->I!=I||d->O!=D) return 0;
|
||||
host[c]={g->weights,u->weights,d->weights,g->scales,u->scales,d->scales,
|
||||
g->fmt,u->fmt,d->fmt,rows[c],total};
|
||||
total+=rows[c];
|
||||
}
|
||||
if(total>8) return 0; /* decode-scale only */
|
||||
DeviceContext *ctx=find_ctx(device); if(!ctx||ctx->group_pending||!select_ctx(ctx)) return 0;
|
||||
size_t xb=(size_t)total*D*sizeof(float), ib=(size_t)total*I*sizeof(float);
|
||||
if(!reserve(&ctx->x,&ctx->x_cap,xb)||!reserve(&ctx->y,&ctx->y_cap,xb)||
|
||||
!reserve(&ctx->gate,&ctx->gate_cap,ib)||!reserve(&ctx->up,&ctx->up_cap,ib)||
|
||||
!reserve_pinned(&ctx->host_x,&ctx->host_x_cap,xb)||
|
||||
!reserve_pinned(&ctx->host_y,&ctx->host_y_cap,xb)) return 0;
|
||||
std::memcpy(ctx->host_x,x,xb);
|
||||
if(!cuda_ok(cudaMemcpyAsync(ctx->x,ctx->host_x,xb,cudaMemcpyHostToDevice,ctx->stream),
|
||||
"expert group issue upload")) return 0;
|
||||
for(int c=0;c<count;c++){
|
||||
int r=rows[c];
|
||||
float *g16=ctx->gate+(size_t)host[c].offset*I,*u16=ctx->up+(size_t)host[c].offset*I;
|
||||
float *x16=ctx->x+(size_t)host[c].offset*D,*y16=ctx->y+(size_t)host[c].offset*D;
|
||||
quant_matmul<<<dim3((unsigned)I,(unsigned)r),256,0,ctx->stream>>>(g16,x16,
|
||||
host[c].g,host[c].gs,host[c].gf,r,D,I,row_bytes(host[c].gf,D),0,1);
|
||||
quant_matmul<<<dim3((unsigned)I,(unsigned)r),256,0,ctx->stream>>>(u16,x16,
|
||||
host[c].u,host[c].us,host[c].uf,r,D,I,row_bytes(host[c].uf,D),0,1);
|
||||
silu_mul<<<(unsigned)(((size_t)r*I+255)/256),256,0,ctx->stream>>>(g16,u16,(size_t)r*I);
|
||||
quant_matmul<<<dim3((unsigned)D,(unsigned)r),256,0,ctx->stream>>>(y16,g16,
|
||||
host[c].d,host[c].ds,host[c].df,r,I,D,row_bytes(host[c].df,I),0,1);
|
||||
}
|
||||
if(!cuda_ok(cudaGetLastError(),"expert group issue launch")||
|
||||
!cuda_ok(cudaMemcpyAsync(ctx->host_y,ctx->y,xb,cudaMemcpyDeviceToHost,ctx->stream),
|
||||
"expert group issue download")) return 0;
|
||||
ctx->group_pending=1; ctx->group_pending_bytes=xb;
|
||||
{ std::lock_guard<std::mutex> lock(g_group_stats_mu);
|
||||
g_group_calls++; g_group_experts+=(uint64_t)count; g_group_rows+=(uint64_t)total; }
|
||||
return 1;
|
||||
}
|
||||
|
||||
extern "C" const float *coli_cuda_expert_group_take(int device) {
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(!ctx||!ctx->group_pending) return nullptr;
|
||||
ctx->group_pending=0;
|
||||
if(!select_ctx(ctx)) return nullptr;
|
||||
if(!cuda_ok(cudaStreamSynchronize(ctx->stream),"expert group take")) return nullptr;
|
||||
return ctx->host_y;
|
||||
}
|
||||
|
||||
|
||||
extern "C" int coli_cuda_attention_absorb(ColiCudaTensor *w,float *ctx,const float *q,
|
||||
const float *latent,const float *rope,int H,int Q,
|
||||
int R,int V,int K,int T,float scale){
|
||||
if (fault_injected()) return 0;
|
||||
if(!w||!ctx||!q||!latent||!rope||H<1||Q<1||R<1||V<1||K<1||K>512||T<1||T>4096||
|
||||
w->I!=K||w->O!=H*(Q+V))return 0;
|
||||
DeviceContext *dc=find_ctx(w->device);if(!select_ctx(dc))return 0;
|
||||
@@ -773,7 +980,7 @@ extern "C" int coli_cuda_attention_absorb(ColiCudaTensor *w,float *ctx,const flo
|
||||
!cuda_ok(cudaMemcpyAsync(dc->ar,rope,rb,cudaMemcpyHostToDevice,dc->stream),"attention rope upload"))return 0;
|
||||
size_t shared=(size_t)(2*K+T)*sizeof(float);
|
||||
attention_absorb_kernel<<<H,256,shared,dc->stream>>>(dc->ac,dc->aq,dc->al,dc->ar,w->weights,w->scales,
|
||||
w->fmt,H,Q,R,V,K,T,scale);
|
||||
w->fmt,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"attention absorb launch")||
|
||||
!cuda_ok(cudaMemcpyAsync(ctx,dc->ac,cb,cudaMemcpyDeviceToHost,dc->stream),"attention context download")||
|
||||
!cuda_ok(cudaStreamSynchronize(dc->stream),"attention synchronize"))return 0;
|
||||
@@ -796,13 +1003,13 @@ static int attention_absorb_batch_run(ColiCudaTensor *w,ColiCudaTensor *proj,flo
|
||||
!cuda_ok(cudaMemcpyAsync(dc->ar,rope,rb,cudaMemcpyHostToDevice,dc->stream),"attention batch rope upload"))return 0;
|
||||
size_t shared=(size_t)(2*K+T+256)*sizeof(float);
|
||||
attention_absorb_batch_kernel<<<dim3(H,S),256,shared,dc->stream>>>(dc->ac,dc->aq,dc->al,
|
||||
dc->ar,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
dc->ar,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"attention batch launch"))return 0;
|
||||
const float *src=dc->ac;size_t ob=cb;
|
||||
if(proj){
|
||||
ob=(size_t)S*proj->O*sizeof(float);if(!reserve(&dc->y,&dc->y_cap,ob))return 0;
|
||||
quant_matmul<<<dim3(proj->O,S),256,0,dc->stream>>>(dc->y,dc->ac,proj->weights,
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I));
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I),proj->gs,proj->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"attention o_proj launch"))return 0;src=dc->y;
|
||||
}
|
||||
if(!cuda_ok(cudaMemcpyAsync(out,src,ob,cudaMemcpyDeviceToHost,dc->stream),
|
||||
@@ -814,12 +1021,14 @@ static int attention_absorb_batch_run(ColiCudaTensor *w,ColiCudaTensor *proj,flo
|
||||
extern "C" int coli_cuda_attention_absorb_batch(ColiCudaTensor *w,float *ctx,const float *q,
|
||||
const float *latent,const float *rope,int S,int H,int Q,int R,int V,int K,int T,
|
||||
float scale){
|
||||
if (fault_injected()) return 0;
|
||||
return attention_absorb_batch_run(w,nullptr,ctx,q,latent,rope,S,H,Q,R,V,K,T,scale);
|
||||
}
|
||||
|
||||
extern "C" int coli_cuda_attention_project_batch(ColiCudaTensor *w,ColiCudaTensor *proj,
|
||||
float *out,const float *q,const float *latent,const float *rope,int S,int H,int Q,
|
||||
int R,int V,int K,int T,float scale){
|
||||
if (fault_injected()) return 0;
|
||||
return attention_absorb_batch_run(w,proj,out,q,latent,rope,S,H,Q,R,V,K,T,scale);
|
||||
}
|
||||
|
||||
@@ -900,7 +1109,7 @@ extern "C" int coli_cuda_attention_project_ragged(ColiCudaTensor *w,ColiCudaTens
|
||||
attention_absorb_ragged_kernel<<<dim3(H,S),256,shared,dc->stream>>>(dc->ac,dc->aq,ddl,ddr,
|
||||
dn,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
quant_matmul<<<dim3(proj->O,S),256,0,dc->stream>>>(dc->y,dc->ac,proj->weights,
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I));
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I),proj->gs,proj->ng);
|
||||
return cuda_ok(cudaGetLastError(),"ragged attention launch")&&
|
||||
cuda_ok(cudaMemcpyAsync(out,dc->y,ob,cudaMemcpyDeviceToHost,dc->stream),"ragged output download")&&
|
||||
cuda_ok(cudaStreamSynchronize(dc->stream),"ragged attention synchronize");
|
||||
@@ -911,7 +1120,8 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) {
|
||||
DeviceContext *ctx = find_ctx(tensor->device);
|
||||
if (ctx) select_ctx(ctx);
|
||||
if (tensor->tracked && ctx) {
|
||||
size_t bytes = tensor->weight_bytes + (tensor->fmt ? (size_t)tensor->O * sizeof(float) : 0);
|
||||
int ng = tensor->ng > 0 ? tensor->ng : 1;
|
||||
size_t bytes = tensor->weight_bytes + (tensor->fmt ? (size_t)tensor->O * ng * sizeof(float) : 0);
|
||||
if (ctx->tensor_count) ctx->tensor_count--;
|
||||
if (ctx->tensor_bytes >= bytes) ctx->tensor_bytes -= bytes;
|
||||
}
|
||||
@@ -925,7 +1135,9 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) {
|
||||
}
|
||||
|
||||
extern "C" size_t coli_cuda_tensor_bytes(const ColiCudaTensor *tensor) {
|
||||
return tensor ? tensor->weight_bytes + (tensor->fmt ? (size_t)tensor->O * sizeof(float) : 0) : 0;
|
||||
if (!tensor) return 0;
|
||||
int ng = tensor->ng > 0 ? tensor->ng : 1;
|
||||
return tensor->weight_bytes + (tensor->fmt ? (size_t)tensor->O * ng * sizeof(float) : 0);
|
||||
}
|
||||
|
||||
extern "C" int coli_cuda_tensor_device(const ColiCudaTensor *tensor) {
|
||||
@@ -986,7 +1198,7 @@ __global__ static void pipe_rows_add(float *x,const float *partial,const int *ro
|
||||
* per layer (78 x ~10 alloc/richiesta erano puro churn). */
|
||||
extern "C" float *coli_cuda_pipe_scratch(int device,int slot,size_t bytes){
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(slot<0||slot>=24||!select_ctx(ctx)) return NULL;
|
||||
if(slot<0||slot>=27||!select_ctx(ctx)) return NULL;
|
||||
if(!reserve(&ctx->pipe_buf[slot],&ctx->pipe_cap[slot],bytes)) return NULL;
|
||||
return ctx->pipe_buf[slot];
|
||||
}
|
||||
@@ -1010,6 +1222,7 @@ extern "C" int coli_cuda_pipe_download(int device,const void *src,void *dst,size
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_rmsnorm(int device,float *y_dev,const float *x_dev,
|
||||
const float *w_dev,int S,int D,float eps){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(S<1||D<1||!select_ctx(ctx)) return 0;
|
||||
pipe_rmsnorm_rows<<<S,256>>>(y_dev,x_dev,w_dev,D,eps,D,D);
|
||||
@@ -1018,6 +1231,7 @@ extern "C" int coli_cuda_pipe_rmsnorm(int device,float *y_dev,const float *x_dev
|
||||
extern "C" int coli_cuda_pipe_rmsnorm_s(int device,float *y_dev,const float *x_dev,
|
||||
const float *w_dev,int S,int D,float eps,
|
||||
int xstride,int ystride){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(S<1||D<1||xstride<D||ystride<D||!select_ctx(ctx)) return 0;
|
||||
pipe_rmsnorm_rows<<<S,256>>>(y_dev,x_dev,w_dev,D,eps,xstride,ystride);
|
||||
@@ -1026,6 +1240,7 @@ extern "C" int coli_cuda_pipe_rmsnorm_s(int device,float *y_dev,const float *x_d
|
||||
extern "C" int coli_cuda_pipe_rope(int device,float *v_dev,const int *pos_dev,
|
||||
int rows,int stride,int offset,int R,int heads,
|
||||
float theta){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(rows<1||R<2||R>256||heads<1||!select_ctx(ctx)) return 0;
|
||||
pipe_rope_rows<<<rows,128>>>(v_dev,pos_dev,0,stride,offset,R,heads,theta);
|
||||
@@ -1033,11 +1248,88 @@ extern "C" int coli_cuda_pipe_rope(int device,float *v_dev,const int *pos_dev,
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_rope_base(int device,float *v_dev,int pos_base,int rows,
|
||||
int stride,int offset,int R,int heads,float theta){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(rows<1||R<2||R>256||heads<1||!select_ctx(ctx)) return 0;
|
||||
pipe_rope_rows<<<rows,128>>>(v_dev,NULL,pos_base,stride,offset,R,heads,theta);
|
||||
return cuda_ok(cudaGetLastError(),"pipe rope base");
|
||||
}
|
||||
/* ---- device router (#431 PR-A) -------------------------------------------
|
||||
* Router for one decode row, entirely on the layer's home device: logits GEMV
|
||||
* (E x D, tiny) + sigmoid, bias-augmented top-K selection, route-level TOPP
|
||||
* truncation, norm_topk and routed_scale — a float-faithful clone of moe()'s
|
||||
* plain routing path (colibri.c FASE A). Selection runs single-thread so the
|
||||
* argmax order, tie-breaking (strict >, lowest index wins) and weight math
|
||||
* match the CPU reference exactly; only the dot/expf rounding can differ,
|
||||
* which is the documented kernel-family divergence class (#100/#163).
|
||||
* Results are packed [idx[K] | w[K] | keff] in one scratch buffer and read
|
||||
* back with a single tiny D2H. */
|
||||
__global__ void pipe_router_logits(const float *__restrict__ x,
|
||||
const float *__restrict__ W,
|
||||
const float *__restrict__ bias,
|
||||
int D, float *logit, float *choice){
|
||||
int e = blockIdx.x;
|
||||
const float *w = W + (size_t)e*D;
|
||||
float acc = 0.f;
|
||||
for(int i=threadIdx.x; i<D; i+=blockDim.x) acc += x[i]*w[i];
|
||||
__shared__ float sh[128];
|
||||
sh[threadIdx.x]=acc; __syncthreads();
|
||||
for(int s=blockDim.x>>1; s>0; s>>=1){
|
||||
if(threadIdx.x<s) sh[threadIdx.x]+=sh[threadIdx.x+s];
|
||||
__syncthreads();
|
||||
}
|
||||
if(!threadIdx.x){
|
||||
float lg = 1.f/(1.f+expf(-sh[0]));
|
||||
logit[e]=lg; choice[e]=lg+bias[e];
|
||||
}
|
||||
}
|
||||
__global__ void pipe_router_select(const float *__restrict__ logit,
|
||||
const float *__restrict__ choice, int E,
|
||||
int Ksel, float topp, int norm_topk,
|
||||
float routed_scale, char *out){
|
||||
if(threadIdx.x||blockIdx.x) return;
|
||||
int *idx = (int*)out;
|
||||
float *w = (float*)(out + Ksel*sizeof(int));
|
||||
int *keff= (int*)(out + Ksel*(sizeof(int)+sizeof(float)));
|
||||
for(int kk=0;kk<Ksel;kk++){
|
||||
int best=-1; float bv=-1e30f;
|
||||
for(int e=0;e<E;e++){ int tk=0; for(int j=0;j<kk;j++) if(idx[j]==e){tk=1;break;}
|
||||
if(!tk && choice[e]>bv){bv=choice[e];best=e;} }
|
||||
idx[kk]=best; w[kk]=logit[best];
|
||||
}
|
||||
int Ke=Ksel;
|
||||
if(topp>0.f && topp<1.f){
|
||||
for(int a=1;a<Ksel;a++){ int ii=idx[a]; float ww=w[a]; int b=a-1;
|
||||
while(b>=0 && w[b]<ww){ w[b+1]=w[b]; idx[b+1]=idx[b]; b--; } w[b+1]=ww; idx[b+1]=ii; }
|
||||
float tot=1e-20f; for(int kk=0;kk<Ksel;kk++) tot+=w[kk];
|
||||
float cum=0.f; for(int kk=0;kk<Ksel;kk++){ cum+=w[kk]; if(cum>=topp*tot){ Ke=kk+1; break; } }
|
||||
}
|
||||
if(norm_topk){ float sm=0.f; for(int kk=0;kk<Ke;kk++) sm+=w[kk]; sm+=1e-20f;
|
||||
for(int kk=0;kk<Ke;kk++) w[kk]/=sm; }
|
||||
for(int kk=0;kk<Ke;kk++) w[kk]*=routed_scale;
|
||||
*keff=Ke;
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_router(int device,const float *x_dev,
|
||||
const void *rw_dev,const void *rb_dev,int D,int E,int Ksel,
|
||||
float topp,int norm_topk,float routed_scale,
|
||||
int *idx_host,float *w_host,int *keff_host){
|
||||
DeviceContext *ctx=find_ctx(device);
|
||||
if(!x_dev||!rw_dev||!rb_dev||D<1||E<1||E>4096||Ksel<1||Ksel>64||!select_ctx(ctx)) return 0;
|
||||
size_t pack=(size_t)Ksel*(sizeof(int)+sizeof(float))+sizeof(int);
|
||||
float *logit=coli_cuda_pipe_scratch(device,22,(size_t)E*sizeof(float));
|
||||
float *chc =coli_cuda_pipe_scratch(device,23,(size_t)E*sizeof(float));
|
||||
char *out =(char*)coli_cuda_pipe_scratch(device,24,pack);
|
||||
if(!logit||!chc||!out) return 0;
|
||||
pipe_router_logits<<<E,128>>>(x_dev,(const float*)rw_dev,(const float*)rb_dev,D,logit,chc);
|
||||
pipe_router_select<<<1,1>>>(logit,chc,E,Ksel,topp,norm_topk,routed_scale,out);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe router launch")) return 0;
|
||||
char buf[64*(sizeof(int)+sizeof(float))+sizeof(int)];
|
||||
if(!cuda_ok(cudaMemcpy(buf,out,pack,cudaMemcpyDeviceToHost),"pipe router readback")) return 0;
|
||||
memcpy(idx_host,buf,(size_t)Ksel*sizeof(int));
|
||||
memcpy(w_host,buf+Ksel*sizeof(int),(size_t)Ksel*sizeof(float));
|
||||
memcpy(keff_host,buf+Ksel*(sizeof(int)+sizeof(float)),sizeof(int));
|
||||
return 1;
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_copy2d(int device,float *dst,int dpitch,const float *src,
|
||||
int spitch,int width,int height){
|
||||
DeviceContext *ctx=find_ctx(device); if(!select_ctx(ctx)) return 0;
|
||||
@@ -1050,6 +1342,7 @@ extern "C" int coli_cuda_pipe_copy2d(int device,float *dst,int dpitch,const floa
|
||||
extern "C" int coli_cuda_attention_project_batch_dev(ColiCudaTensor *w,ColiCudaTensor *proj,
|
||||
float *out,const float *q_dev,const float *latent_dev,const float *rope_dev,
|
||||
int S,int H,int Q,int R,int V,int K,int T,float scale){
|
||||
if (fault_injected()) return 0;
|
||||
if(!w||!proj||!out||!q_dev||!latent_dev||!rope_dev||S<1||H<1||Q<1||R<1||V<1||
|
||||
K<1||K>512||T<S||T>8192||w->I!=K||w->O!=H*(Q+V)||
|
||||
proj->device!=w->device||proj->I!=H*V)return 0;
|
||||
@@ -1058,12 +1351,12 @@ extern "C" int coli_cuda_attention_project_batch_dev(ColiCudaTensor *w,ColiCudaT
|
||||
if(!reserve(&dc->ac,&dc->ac_cap,cb))return 0;
|
||||
size_t shared=(size_t)(2*K+T+256)*sizeof(float);
|
||||
attention_absorb_batch_kernel<<<dim3(H,S),256,shared,dc->stream>>>(dc->ac,q_dev,latent_dev,
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe attention launch"))return 0;
|
||||
size_t ob=(size_t)S*proj->O*sizeof(float);
|
||||
if(!reserve(&dc->y,&dc->y_cap,ob))return 0;
|
||||
quant_matmul<<<dim3(proj->O,S),256,0,dc->stream>>>(dc->y,dc->ac,proj->weights,
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I));
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I),proj->gs,proj->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe o_proj launch"))return 0;
|
||||
if(!cuda_ok(cudaMemcpyAsync(out,dc->y,ob,cudaMemcpyDeviceToHost,dc->stream),"pipe attention download")||
|
||||
!cuda_ok(cudaStreamSynchronize(dc->stream),"pipe attention sync"))return 0;
|
||||
@@ -1071,17 +1364,20 @@ extern "C" int coli_cuda_attention_project_batch_dev(ColiCudaTensor *w,ColiCudaT
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_silu_mul(int device,float *gate_dev,const float *up_dev,
|
||||
size_t n){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device); if(!n||!select_ctx(ctx)) return 0;
|
||||
silu_mul<<<(unsigned)((n+255)/256),256>>>(gate_dev,up_dev,n);
|
||||
return cuda_ok(cudaGetLastError(),"pipe silu mul");
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_add(int device,float *x_dev,const float *t_dev,size_t n){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device); if(!n||!select_ctx(ctx)) return 0;
|
||||
pipe_add_n<<<(unsigned)((n+255)/256),256>>>(x_dev,t_dev,n);
|
||||
return cuda_ok(cudaGetLastError(),"pipe add");
|
||||
}
|
||||
extern "C" int coli_cuda_pipe_rows_add(int device,float *x_dev,const float *partial_dev,
|
||||
const int *rows_dev,int nrows,int D){
|
||||
if (fault_injected()) return 0;
|
||||
DeviceContext *ctx=find_ctx(device); if(nrows<1||D<1||!select_ctx(ctx)) return 0;
|
||||
pipe_rows_add<<<nrows,256>>>(x_dev,partial_dev,rows_dev,D);
|
||||
return cuda_ok(cudaGetLastError(),"pipe rows add");
|
||||
@@ -1090,11 +1386,12 @@ extern "C" int coli_cuda_pipe_rows_add(int device,float *x_dev,const float *part
|
||||
* coli_cuda_matmul, zero host transfers. */
|
||||
extern "C" int coli_cuda_pipe_gemm(ColiCudaTensor *t,float *y_dev,const float *x_dev,
|
||||
int S){
|
||||
if (fault_injected()) return 0;
|
||||
if(!t||S<1) return 0;
|
||||
DeviceContext *ctx=find_ctx(t->device); if(!select_ctx(ctx)) return 0;
|
||||
dim3 grid((unsigned)t->O,(unsigned)S);
|
||||
quant_matmul<<<grid,256>>>(y_dev,x_dev,t->weights,t->scales,t->fmt,S,t->I,t->O,
|
||||
row_bytes(t->fmt,t->I));
|
||||
row_bytes(t->fmt,t->I),t->gs,t->ng);
|
||||
return cuda_ok(cudaGetLastError(),"pipe gemm");
|
||||
}
|
||||
/* copia diretta scheda->scheda (P2P se disponibile, altrimenti staging driver) */
|
||||
@@ -1109,6 +1406,7 @@ extern "C" int coli_cuda_pipe_peer_copy(int dst_dev,float *dst,int src_dev,
|
||||
extern "C" int coli_cuda_attention_project_batch_dev_out(ColiCudaTensor *w,ColiCudaTensor *proj,
|
||||
float *out_dev,const float *q_dev,const float *latent_dev,const float *rope_dev,
|
||||
int S,int H,int Q,int R,int V,int K,int T,float scale){
|
||||
if (fault_injected()) return 0;
|
||||
if(!w||!proj||!out_dev||!q_dev||!latent_dev||!rope_dev||S<1||H<1||Q<1||R<1||V<1||
|
||||
K<1||K>512||T<S||T>8192||w->I!=K||w->O!=H*(Q+V)||
|
||||
proj->device!=w->device||proj->I!=H*V)return 0;
|
||||
@@ -1117,10 +1415,10 @@ extern "C" int coli_cuda_attention_project_batch_dev_out(ColiCudaTensor *w,ColiC
|
||||
if(!reserve(&dc->ac,&dc->ac_cap,cb))return 0;
|
||||
size_t shared=(size_t)(2*K+T+256)*sizeof(float);
|
||||
attention_absorb_batch_kernel<<<dim3(H,S),256,shared,dc->stream>>>(dc->ac,q_dev,latent_dev,
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe attention launch (dev out)"))return 0;
|
||||
quant_matmul<<<dim3(proj->O,S),256,0,dc->stream>>>(out_dev,dc->ac,proj->weights,
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I));
|
||||
proj->scales,proj->fmt,S,proj->I,proj->O,row_bytes(proj->fmt,proj->I),proj->gs,proj->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe o_proj launch (dev out)"))return 0;
|
||||
return cuda_ok(cudaStreamSynchronize(dc->stream),"pipe attention sync (dev out)");
|
||||
}
|
||||
@@ -1130,12 +1428,13 @@ extern "C" int coli_cuda_attention_project_batch_dev_out(ColiCudaTensor *w,ColiC
|
||||
extern "C" int coli_cuda_attention_absorb_batch_dev(ColiCudaTensor *w,float *ctx_dev,
|
||||
const float *q_dev,const float *latent_dev,const float *rope_dev,
|
||||
int S,int H,int Q,int R,int V,int K,int T,float scale){
|
||||
if (fault_injected()) return 0;
|
||||
if(!w||!ctx_dev||!q_dev||!latent_dev||!rope_dev||S<1||H<1||Q<1||R<1||V<1||
|
||||
K<1||K>512||T<S||T>8192||w->I!=K||w->O!=H*(Q+V))return 0;
|
||||
DeviceContext *dc=find_ctx(w->device);if(!select_ctx(dc))return 0;
|
||||
size_t shared=(size_t)(2*K+T+256)*sizeof(float);
|
||||
attention_absorb_batch_kernel<<<dim3(H,S),256,shared,dc->stream>>>(ctx_dev,q_dev,latent_dev,
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale);
|
||||
rope_dev,w->weights,w->scales,w->fmt,S,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"pipe shard attention launch"))return 0;
|
||||
return cuda_ok(cudaStreamSynchronize(dc->stream),"pipe shard attention sync");
|
||||
}
|
||||
@@ -1144,6 +1443,7 @@ extern "C" int coli_cuda_attention_absorb_batch_dev(ColiCudaTensor *w,float *ctx
|
||||
extern "C" int coli_cuda_attention_absorb_kvdev(ColiCudaTensor *w,float *ctx,const float *q,
|
||||
const float *latent_dev,const float *rope_dev,int H,int Q,int R,int V,int K,int T,
|
||||
float scale){
|
||||
if (fault_injected()) return 0;
|
||||
if(!w||!ctx||!q||!latent_dev||!rope_dev||H<1||Q<1||R<1||V<1||K<1||K>512||T<1||T>8192||
|
||||
w->I!=K||w->O!=H*(Q+V))return 0;
|
||||
DeviceContext *dc=find_ctx(w->device);if(!select_ctx(dc))return 0;
|
||||
@@ -1152,7 +1452,7 @@ extern "C" int coli_cuda_attention_absorb_kvdev(ColiCudaTensor *w,float *ctx,con
|
||||
if(!cuda_ok(cudaMemcpyAsync(dc->aq,q,qb,cudaMemcpyHostToDevice,dc->stream),"kvdev q upload"))return 0;
|
||||
size_t shared=(size_t)(2*K+T+256)*sizeof(float);
|
||||
attention_absorb_batch_kernel<<<dim3(H,1),256,shared,dc->stream>>>(dc->ac,dc->aq,latent_dev,
|
||||
rope_dev,w->weights,w->scales,w->fmt,1,H,Q,R,V,K,T,scale);
|
||||
rope_dev,w->weights,w->scales,w->fmt,1,H,Q,R,V,K,T,scale,w->gs,w->ng);
|
||||
if(!cuda_ok(cudaGetLastError(),"kvdev absorb launch")||
|
||||
!cuda_ok(cudaMemcpyAsync(ctx,dc->ac,cb,cudaMemcpyDeviceToHost,dc->stream),"kvdev ctx download")||
|
||||
!cuda_ok(cudaStreamSynchronize(dc->stream),"kvdev absorb sync"))return 0;
|
||||
|
||||
+21
-3
@@ -36,20 +36,24 @@ COLI_CUDA_DLLEXPORT void coli_cuda_group_stats(uint64_t *calls, uint64_t *expert
|
||||
double *h2d_ms, double *kernel_ms, double *d2h_ms);
|
||||
|
||||
/* Upload without executing, so capacity failures happen during model startup. */
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_tensor_upload_g(ColiCudaTensor **tensor,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int I, int O, int device, int gs);
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_tensor_upload(ColiCudaTensor **tensor,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int I, int O, int device);
|
||||
|
||||
/*
|
||||
* y[S,O] = x[S,I] @ W[O,I]^T.
|
||||
* fmt matches QT in glm.c: 0=f32, 1=int8, 2=int4, 3=int2.
|
||||
* The first successful call uploads W and its row scales; later calls reuse it.
|
||||
* fmt matches QT in glm.c: 0=f32, 1=int8, 2=int4, 3=int2, 4=grouped int4.
|
||||
* gs is the group size for fmt=4 (0 for all other formats).
|
||||
* The first successful call uploads W and its scales; later calls reuse it.
|
||||
* Returns 1 on success and 0 when CUDA is not initialized or the format is invalid.
|
||||
*/
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_matmul(ColiCudaTensor **tensor,
|
||||
float *y, const float *x,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int S, int I, int O, int device);
|
||||
int fmt, int S, int I, int O, int device, int gs);
|
||||
|
||||
/* Fused expert pipeline: y = down(silu(gate(x)) * up(x)). All three tensors
|
||||
* must already be resident on one device. Activations cross PCIe once in
|
||||
@@ -67,6 +71,16 @@ COLI_CUDA_DLLEXPORT int coli_cuda_shared_mlp_w4a16(ColiCudaTensor *gate, ColiCud
|
||||
|
||||
/* Packed group of same-shaped experts. Inputs and outputs contain sum(rows)
|
||||
* consecutive [D] rows in call order. */
|
||||
/* Async issue/take split of the group call below (Inc.4): issue launches on the
|
||||
* device stream and returns; take syncs and returns the pinned result rows (valid
|
||||
* until the next issue on that device). Small totals only (<=8 rows); one
|
||||
* outstanding issue per device. */
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_expert_group_issue(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs,
|
||||
const int *rows, int count, const float *x);
|
||||
COLI_CUDA_DLLEXPORT const float *coli_cuda_expert_group_take(int device);
|
||||
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_expert_group(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs,
|
||||
@@ -126,6 +140,10 @@ COLI_CUDA_DLLEXPORT int coli_cuda_pipe_rmsnorm_s(int device,float *y_dev,const f
|
||||
int xstride,int ystride);
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_pipe_rope_base(int device,float *v_dev,int pos_base,int rows,
|
||||
int stride,int offset,int R,int heads,float theta);
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_pipe_router(int device,const float *x_dev,
|
||||
const void *rw_dev,const void *rb_dev,int D,int E,int Ksel,
|
||||
float topp,int norm_topk,float routed_scale,
|
||||
int *idx_host,float *w_host,int *keff_host);
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_pipe_copy2d(int device,float *dst,int dpitch,const float *src,
|
||||
int spitch,int width,int height);
|
||||
COLI_CUDA_DLLEXPORT int coli_cuda_attention_project_batch_dev(ColiCudaTensor *kv_b,ColiCudaTensor *o_proj,
|
||||
|
||||
+41
-3
@@ -41,14 +41,20 @@ typedef int (*fn_expert_mlp)(ColiCudaTensor *gate, ColiCudaTensor *up
|
||||
typedef int (*fn_expert_group)(ColiCudaTensor *const *gates, ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs, const int *rows, int count,
|
||||
float *y, const float *x);
|
||||
typedef int (*fn_expert_group_issue)(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs,
|
||||
const int *rows, int count, const float *x);
|
||||
typedef const float * (*fn_expert_group_take)(int device);
|
||||
typedef int (*fn_attention_absorb)(ColiCudaTensor *kv_b, float *ctx, const float *q,
|
||||
const float *latent, const float *rope, int H, int Q,
|
||||
int R, int V, int K, int T, float attention_scale);
|
||||
typedef int (*fn_tensor_upload)(ColiCudaTensor **tensor, const void *weights,
|
||||
const float *scales, int fmt, int I, int O, int device);
|
||||
typedef int (*fn_tensor_upload_g)(ColiCudaTensor **tensor, const void *weights, const float *scales, int fmt, int I, int O, int device, int gs);
|
||||
typedef int (*fn_matmul)(ColiCudaTensor **tensor, float *y, const float *x,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int S, int I, int O, int device);
|
||||
int fmt, int S, int I, int O, int device, int gs);
|
||||
typedef void (*fn_tensor_free)(ColiCudaTensor *tensor);
|
||||
typedef size_t (*fn_tensor_bytes)(const ColiCudaTensor *tensor);
|
||||
typedef int (*fn_tensor_device)(const ColiCudaTensor *tensor);
|
||||
@@ -76,6 +82,7 @@ typedef int (*fn_pipe_gemm)(ColiCudaTensor *t,float *y_dev,const float *x_dev,in
|
||||
typedef int (*fn_pipe_peer_copy)(int dst_dev,float *dst,int src_dev, const float *src,size_t bytes);
|
||||
typedef int (*fn_pipe_rmsnorm)(int device,float *y_dev,const float *x_dev, const float *w_dev,int S,int D,float eps);
|
||||
typedef int (*fn_pipe_rmsnorm_s)(int device,float *y_dev,const float *x_dev, const float *w_dev,int S,int D,float eps, int xstride,int ystride);
|
||||
typedef int (*fn_pipe_router)(int device,const float *x_dev,const void *rw_dev,const void *rb_dev,int D,int E,int Ksel,float topp,int norm_topk,float routed_scale,int *idx_host,float *w_host,int *keff_host);
|
||||
typedef int (*fn_pipe_rope)(int device,float *v_dev,const int *pos_dev,int rows, int stride,int offset,int R,int heads,float theta);
|
||||
typedef int (*fn_pipe_rope_base)(int device,float *v_dev,int pos_base,int rows, int stride,int offset,int R,int heads,float theta);
|
||||
typedef int (*fn_pipe_rows_add)(int device,float *x_dev,const float *partial_dev, const int *rows_dev,int nrows,int D);
|
||||
@@ -100,8 +107,11 @@ static struct {
|
||||
fn_group_stats group_stats;
|
||||
fn_expert_mlp expert_mlp;
|
||||
fn_expert_group expert_group;
|
||||
fn_expert_group_issue expert_group_issue;
|
||||
fn_expert_group_take expert_group_take;
|
||||
fn_attention_absorb attention_absorb;
|
||||
fn_tensor_upload tensor_upload;
|
||||
fn_tensor_upload_g tensor_upload_g;
|
||||
fn_matmul matmul;
|
||||
fn_tensor_free tensor_free;
|
||||
fn_tensor_bytes tensor_bytes;
|
||||
@@ -123,6 +133,7 @@ static struct {
|
||||
fn_pipe_peer_copy pipe_peer_copy;
|
||||
fn_pipe_rmsnorm pipe_rmsnorm;
|
||||
fn_pipe_rmsnorm_s pipe_rmsnorm_s;
|
||||
fn_pipe_router pipe_router;
|
||||
fn_pipe_rope pipe_rope;
|
||||
fn_pipe_rope_base pipe_rope_base;
|
||||
fn_pipe_rows_add pipe_rows_add;
|
||||
@@ -194,8 +205,11 @@ static int coli_cuda_load(void){
|
||||
RESOLVE(group_stats, fn_group_stats)
|
||||
RESOLVE(expert_mlp, fn_expert_mlp)
|
||||
RESOLVE(expert_group, fn_expert_group)
|
||||
RESOLVE(expert_group_issue, fn_expert_group_issue)
|
||||
RESOLVE(expert_group_take, fn_expert_group_take)
|
||||
RESOLVE(attention_absorb, fn_attention_absorb)
|
||||
RESOLVE(tensor_upload, fn_tensor_upload)
|
||||
RESOLVE(tensor_upload_g, fn_tensor_upload_g)
|
||||
RESOLVE(matmul, fn_matmul)
|
||||
RESOLVE(tensor_free, fn_tensor_free)
|
||||
RESOLVE(tensor_bytes, fn_tensor_bytes)
|
||||
@@ -217,6 +231,7 @@ static int coli_cuda_load(void){
|
||||
RESOLVE(pipe_peer_copy, fn_pipe_peer_copy)
|
||||
RESOLVE(pipe_rmsnorm, fn_pipe_rmsnorm)
|
||||
RESOLVE(pipe_rmsnorm_s, fn_pipe_rmsnorm_s)
|
||||
RESOLVE(pipe_router, fn_pipe_router)
|
||||
RESOLVE(pipe_rope, fn_pipe_rope)
|
||||
RESOLVE(pipe_rope_base, fn_pipe_rope_base)
|
||||
RESOLVE(pipe_rows_add, fn_pipe_rows_add)
|
||||
@@ -289,6 +304,19 @@ int coli_cuda_expert_group(ColiCudaTensor *const *gates, ColiCudaTensor *const *
|
||||
return g_cuda.expert_group(gates, ups, downs, rows, count, y, x);
|
||||
}
|
||||
|
||||
int coli_cuda_expert_group_issue(ColiCudaTensor *const *gates,
|
||||
ColiCudaTensor *const *ups,
|
||||
ColiCudaTensor *const *downs,
|
||||
const int *rows, int count, const float *x){
|
||||
if(!g_cuda.available) return 0;
|
||||
return g_cuda.expert_group_issue(gates, ups, downs, rows, count, x);
|
||||
}
|
||||
|
||||
const float *coli_cuda_expert_group_take(int device){
|
||||
if(!g_cuda.available) return NULL;
|
||||
return g_cuda.expert_group_take(device);
|
||||
}
|
||||
|
||||
int coli_cuda_attention_absorb(ColiCudaTensor *kv_b, float *ctx, const float *q,
|
||||
const float *latent, const float *rope, int H, int Q,
|
||||
int R, int V, int K, int T, float attention_scale){
|
||||
@@ -302,11 +330,16 @@ int coli_cuda_tensor_upload(ColiCudaTensor **tensor, const void *weights,
|
||||
return g_cuda.tensor_upload(tensor, weights, scales, fmt, I, O, device);
|
||||
}
|
||||
|
||||
int coli_cuda_tensor_upload_g(ColiCudaTensor **tensor, const void *weights, const float *scales, int fmt, int I, int O, int device, int gs){
|
||||
if(!g_cuda.available || !g_cuda.tensor_upload_g){ return 0; }
|
||||
return g_cuda.tensor_upload_g(tensor, weights, scales, fmt, I, O, device, gs);
|
||||
}
|
||||
|
||||
int coli_cuda_matmul(ColiCudaTensor **tensor, float *y, const float *x,
|
||||
const void *weights, const float *scales,
|
||||
int fmt, int S, int I, int O, int device){
|
||||
int fmt, int S, int I, int O, int device, int gs){
|
||||
if(!g_cuda.available) return 0;
|
||||
return g_cuda.matmul(tensor, y, x, weights, scales, fmt, S, I, O, device);
|
||||
return g_cuda.matmul(tensor, y, x, weights, scales, fmt, S, I, O, device, gs);
|
||||
}
|
||||
|
||||
void coli_cuda_tensor_free(ColiCudaTensor *tensor){
|
||||
@@ -407,6 +440,11 @@ int coli_cuda_pipe_rmsnorm(int device,float *y_dev,const float *x_dev, const flo
|
||||
return g_cuda.pipe_rmsnorm(device, y_dev, x_dev, w_dev, S, D, eps);
|
||||
}
|
||||
|
||||
int coli_cuda_pipe_router(int device,const float *x_dev,const void *rw_dev,const void *rb_dev,int D,int E,int Ksel,float topp,int norm_topk,float routed_scale,int *idx_host,float *w_host,int *keff_host){
|
||||
if(!g_cuda.available || !g_cuda.pipe_router){ return 0; }
|
||||
return g_cuda.pipe_router(device, x_dev, rw_dev, rb_dev, D, E, Ksel, topp, norm_topk, routed_scale, idx_host, w_host, keff_host);
|
||||
}
|
||||
|
||||
int coli_cuda_pipe_rmsnorm_s(int device,float *y_dev,const float *x_dev, const float *w_dev,int S,int D,float eps, int xstride,int ystride){
|
||||
if(!g_cuda.available){ return 0; }
|
||||
return g_cuda.pipe_rmsnorm_s(device, y_dev, x_dev, w_dev, S, D, eps, xstride, ystride);
|
||||
|
||||
@@ -84,6 +84,23 @@ int coli_metal_layer_decode(float *x,
|
||||
|
||||
int coli_metal_gemm(float *y, const float *x, const void *weights, const float *scales,
|
||||
int fmt, int S, int I, int O); /* large-batch sync GEMM; 0 -> CPU */
|
||||
/* Parallel top-8 expert selection (r_top8_par): run ONE top-8 selection kernel standalone
|
||||
* on host arrays — par=0 the serial r_top8, par=1 the parallel exact-match replica gated
|
||||
* in the engine by COLI_RTOP8 (default ON; COLI_RTOP8=0 opts out to the serial kernel).
|
||||
* Exists so the metal-test suite (and any battery probe) can prove serial/parallel
|
||||
* equivalence on the ENGINE build's own compiled shaders, not just in the bench tool.
|
||||
* sig[S*E], bias[E], idx[S*K], w[S*K], keff[S].
|
||||
* Expert-count generality: the parallel kernel's blocked-lane design (ch[8]/32-lane
|
||||
* threadgroup) is validated correct for arbitrary E<=256, including non-multiples of the
|
||||
* 32-lane width and small E (see metal-test's E=24/E=168/E=256 cases — 168 is the REAP
|
||||
* expert-pruned package width from #428/#426). For E>256 (out of contract) this function
|
||||
* transparently falls back to the serial kernel even when par=1 is requested, and the
|
||||
* same automatic fallback is wired into the engine dispatch site — "par" is a request,
|
||||
* never a guarantee, so no caller can reach the unguarded parallel path out of contract.
|
||||
* Returns 1 on success, 0 if Metal is unavailable. */
|
||||
int coli_metal_rtop8(int par, const float *sig, const float *bias, int S, int E, int K,
|
||||
int Ksel, float topp, int normk, float rscale,
|
||||
int *idx, float *w, int *keff);
|
||||
void coli_metal_attn_counts(uint64_t *ok, double *wall, double *kernel);
|
||||
void coli_metal_attn_lat(double *ksched, double *gsched);
|
||||
int coli_metal_attn_decode(const float *x,
|
||||
@@ -98,6 +115,10 @@ int coli_metal_attn_decode(const float *x,
|
||||
void coli_metal_moe_counts(uint64_t *ok, uint64_t *fb, uint64_t *experts);
|
||||
void coli_metal_moe_times(double *setup, double *gpu, double *scatter);
|
||||
double coli_metal_moe_kernel_time(void);
|
||||
/* E5 (COLI_METAL_RESSET=1): returns 1 when the queue-attached residency set is active and
|
||||
* writes the cumulative seconds moe_submit spent committing pending set adds -- a cost that
|
||||
* sits OUTSIDE the setup/gpu/scatter breakdown above. Returns 0 (and writes 0) when off. */
|
||||
int coli_metal_resset_stats(double *flush_s);
|
||||
|
||||
/*
|
||||
* Batched routed-expert SwiGLU for one MoE block, in ONE command buffer.
|
||||
|
||||
+265
-12
@@ -219,6 +219,63 @@ kernel void r_top8(device const float* sig [[buffer(0)]], device const float* bi
|
||||
if(normk){ float sm=0; for(int kk=0;kk<Ke;kk++) sm+=ww[kk]; sm+=1e-20f; for(int kk=0;kk<Ke;kk++) ww[kk]/=sm; }
|
||||
for(int kk=0;kk<Ke;kk++) ww[kk]*=rscale;
|
||||
}
|
||||
// parallel replica of r_top8's selection on ONE SIMDGROUP per row instead of one serial
|
||||
// thread (bench/kernels @ 27bfe83: serial r_top8 measured 0.465 ms/layer, ~55% of the
|
||||
// layer CB; this replica ~93x faster with exactly matching output). EXACT-MATCH is the
|
||||
// contract: each lane owns ceil(E/32) contiguous experts (blocked) and keeps a taken
|
||||
// bitmask; per selection step: lane-local strict-'>' ascending max (lowest index wins
|
||||
// within a lane, matching the serial ascending scan), then a shuffle-down argmax
|
||||
// reduction where ties prefer the LOWER index — together exactly the serial kernel's
|
||||
// first-max-wins order. The topp/normk/rscale tail is the serial code verbatim on lane 0
|
||||
// (same ops, same order => bitwise-identical results; metal-test enforces this with
|
||||
// memcmp). Contract: E<=256 (ch[8]/taken mask sizing: ceil(E/32)<=8) — the defensive
|
||||
// return below makes an out-of-contract dispatch a visible no-op (idx/w/keff untouched),
|
||||
// never an OOB write; both call sites (coli_metal_layer_decode's dispatch and the
|
||||
// standalone coli_metal_rtop8 runner) additionally gate on E<=256 in host code before
|
||||
// selecting this pipeline at all, so the return here is defense-in-depth, not the only
|
||||
// guard. Sentinel-per-lane design (ch[j]=-1e30f for e>=E) makes non-multiple-of-32 E
|
||||
// and small E correct without special-casing — validated for E=24, E=168 (REAP
|
||||
// expert-pruned packages, see the upstream feature-request thread) and E=256 by metal-test.
|
||||
// ASSUMES SIMD width 32 (shuffle offsets 16..1, 32-thread threadgroup per row): enforced
|
||||
// at init — coli_metal_init clears g_rtop8_width_ok (and therefore both call sites' use
|
||||
// of this pipeline) if threadExecutionWidth != 32.
|
||||
kernel void r_top8_par(device const float* sig [[buffer(0)]], device const float* bias [[buffer(1)]],
|
||||
device int* idx [[buffer(2)]], device float* w [[buffer(3)]],
|
||||
device int* keff [[buffer(4)]], constant int& E [[buffer(5)]],
|
||||
constant int& K [[buffer(6)]], constant int& Ksel [[buffer(7)]],
|
||||
constant float& topp [[buffer(8)]], constant int& normk [[buffer(9)]],
|
||||
constant float& rscale [[buffer(10)]],
|
||||
uint s [[threadgroup_position_in_grid]],
|
||||
uint slane [[thread_index_in_simdgroup]]) {
|
||||
if(E>256) return;
|
||||
device const float* sg=sig+(long)s*E;
|
||||
device int* id_=idx+(long)s*K; device float* ww=w+(long)s*K;
|
||||
int per=(E+31)/32, base=(int)slane*per;
|
||||
float ch[8]; uint taken=0u;
|
||||
for(int j=0;j<per;j++){ int e=base+j; ch[j]=(e<E)?sg[e]+bias[e]:-1e30f; }
|
||||
for(int kk=0;kk<Ksel;kk++){
|
||||
float bv=-1e30f; int bi=0x7FFFFFFF;
|
||||
for(int j=0;j<per;j++) if(!(taken&(1u<<j)) && ch[j]>bv){ bv=ch[j]; bi=base+j; }
|
||||
for(uint off=16;off>0;off>>=1){
|
||||
float ov=simd_shuffle_down(bv,off); int oi=simd_shuffle_down(bi,off);
|
||||
if(ov>bv || (ov==bv && oi<bi)){ bv=ov; bi=oi; }
|
||||
}
|
||||
bv=simd_broadcast(bv,0); bi=simd_broadcast(bi,0);
|
||||
if(bi>=base && bi<base+per) taken|=1u<<(bi-base);
|
||||
if(slane==0){ id_[kk]=bi; ww[kk]=sg[bi]; }
|
||||
}
|
||||
if(slane!=0) return;
|
||||
int Ke=Ksel;
|
||||
if(topp>0.0f && topp<1.0f){
|
||||
for(int a=1;a<Ksel;a++){ int ii=id_[a]; float wv=ww[a]; int b=a-1;
|
||||
while(b>=0 && ww[b]<wv){ ww[b+1]=ww[b]; id_[b+1]=id_[b]; b--; } ww[b+1]=wv; id_[b+1]=ii; }
|
||||
float tot=1e-20f; for(int kk=0;kk<Ksel;kk++) tot+=ww[kk];
|
||||
float cum=0; for(int kk=0;kk<Ksel;kk++){ cum+=ww[kk]; if(cum>=topp*tot){ Ke=kk+1; break; } }
|
||||
}
|
||||
keff[s]=Ke;
|
||||
if(normk){ float sm=0; for(int kk=0;kk<Ke;kk++) sm+=ww[kk]; sm+=1e-20f; for(int kk=0;kk<Ke;kk++) ww[kk]/=sm; }
|
||||
for(int kk=0;kk<Ke;kk++) ww[kk]*=rscale;
|
||||
}
|
||||
)METAL";
|
||||
|
||||
struct ColiMetalTensor {
|
||||
@@ -231,7 +288,15 @@ static id<MTLDevice> g_dev;
|
||||
static id<MTLCommandQueue> g_queue;
|
||||
static id<MTLComputePipelineState> g_gemv, g_moe_gemv, g_moe_silu;
|
||||
static id<MTLComputePipelineState> g_a_rms, g_a_rope, g_a_copy, g_a_qabs, g_a_score, g_a_smax, g_a_clat, g_a_ctx;
|
||||
static id<MTLComputePipelineState> g_a_add, g_r_router, g_r_top8;
|
||||
static id<MTLComputePipelineState> g_a_add, g_r_router, g_r_top8, g_r_top8p;
|
||||
static int g_rtop8_par = 1; // COLI_RTOP8 (default ON); COLI_RTOP8=0 opts out to the
|
||||
// serial kernel — see coli_metal_init.
|
||||
static int g_rtop8_width_ok = 1; // hardware fact, independent of the policy gate above:
|
||||
// false if this device's threadExecutionWidth != 32.
|
||||
// Consulted by BOTH the engine dispatch site and the
|
||||
// standalone coli_metal_rtop8 runner, so no caller can
|
||||
// reach r_top8_par's 32-lane reduction on an unsafe
|
||||
// device even by explicitly requesting par=1.
|
||||
static size_t g_tensor_count, g_tensor_bytes;
|
||||
static uint64_t g_moe_ok, g_moe_fb, g_moe_experts; // GPU blocks / CPU-fallback blocks / experts on GPU
|
||||
static double g_t_setup, g_t_gpu, g_t_scatter, g_t_kernel; // per-block time breakdown (seconds)
|
||||
@@ -258,6 +323,73 @@ extern "C" void coli_metal_attn_lat(double *ksched, double *gsched){
|
||||
struct Slab { void *base; size_t len; id<MTLBuffer> buf; };
|
||||
static std::vector<Slab> g_slabs;
|
||||
static std::mutex g_slab_mtx; // expert_load registers slabs from parallel OpenMP threads
|
||||
|
||||
// ---- E5 experiment: COLI_METAL_RESSET=1 -- one persistent MTLResidencySet attached to
|
||||
// g_queue (macOS 15+) replaces moe_submit's per-command-buffer useResource: loop over
|
||||
// resolved expert weight/scale slabs. Allocation is untouched (same newBufferWithBytesNoCopy
|
||||
// wrap as stock); only residency bookkeeping moves off the dispatch hot path -- see
|
||||
// SUMMARY.md for why skipping useResource: there is safe (read-only, indirectly-referenced
|
||||
// buffers only; residency sets don't do hazard tracking, but nothing here relied on it).
|
||||
// g_resset_obj is a bare `id` (holds id<MTLResidencySet>) so the global's declared type
|
||||
// carries no availability annotation -- the protocol name only appears inside
|
||||
// @available(macOS 15.0, *) guards below, keeping -Wunguarded-availability clean.
|
||||
static id g_resset_obj;
|
||||
static bool g_resset_enabled; // COLI_METAL_RESSET=1, macOS 15+, and creation succeeded
|
||||
static bool g_resset_dirty; // addAllocation: calls pending commit; g_resset_mtx-guarded
|
||||
// Set mutations + dirty flag get their OWN mutex, never held together with g_slab_mtx: no
|
||||
// live Metal call may run under the slab lock the parallel OMP loader threads contend on
|
||||
// (E4's audit round 2 found exactly that shape -- mutex over a live Metal call -- as the
|
||||
// leading suspect for its +12s expert-disk regression). g_slab_mtx keeps guarding g_slabs
|
||||
// bookkeeping only, exactly as on stock.
|
||||
static std::mutex g_resset_mtx;
|
||||
static double g_t_resset_flush; // sec committing pending adds in moe_submit (gate on only)
|
||||
|
||||
// Add a just-wrapped buffer to the set; commit deferred (an OMP loader burst batches into
|
||||
// one commit at the next moe_submit instead of one per slab). Called by coli_metal_register
|
||||
// after it drops g_slab_mtx but before it returns -- and the engine cannot dispatch an
|
||||
// expert before the load that registers its slab returns, so any slab a given moe_submit
|
||||
// can resolve() was added (and marked dirty) under g_resset_mtx strictly before that
|
||||
// moe_submit's resset_flush() acquired the same mutex: the flush covers it. The slab-table
|
||||
// ordering itself (register-before-resolve) is unchanged and stays under g_slab_mtx.
|
||||
// Cost lands in the caller's existing expert-load accounting (t_ewait window in colibri.c);
|
||||
// no separate counter for the add/remove side.
|
||||
static void resset_add(id<MTLBuffer> b) {
|
||||
if (!g_resset_enabled) return;
|
||||
std::lock_guard<std::mutex> lk(g_resset_mtx);
|
||||
if (@available(macOS 15.0, *)) { [(id<MTLResidencySet>)g_resset_obj addAllocation:b]; g_resset_dirty = true; }
|
||||
}
|
||||
// Remove + commit immediately, NOT deferred: the caller frees the underlying host memory
|
||||
// right after coli_metal_unregister returns, so the removal must be applied before that --
|
||||
// an uncommitted-but-still-resident allocation pointing at freed memory is a use-after-free
|
||||
// risk the GPU could act on. Also runs outside g_slab_mtx (see g_resset_mtx above).
|
||||
static void resset_remove(id<MTLBuffer> b) {
|
||||
if (!g_resset_enabled) return;
|
||||
std::lock_guard<std::mutex> lk(g_resset_mtx);
|
||||
if (@available(macOS 15.0, *)) {
|
||||
id<MTLResidencySet> rs = (id<MTLResidencySet>)g_resset_obj;
|
||||
[rs removeAllocation:b]; [rs commit];
|
||||
}
|
||||
g_resset_dirty = false; // commit above also flushes any pending adds
|
||||
}
|
||||
// Flush pending adds before moe_submit relies on the set alone for residency -- the only
|
||||
// caller that skips per-buffer useResource: (see moe_submit below). Takes g_resset_mtx
|
||||
// only, never g_slab_mtx; the happens-before argument lives at resset_add above.
|
||||
static void resset_flush() {
|
||||
if (!g_resset_enabled) return;
|
||||
std::lock_guard<std::mutex> lk(g_resset_mtx);
|
||||
if (!g_resset_dirty) return;
|
||||
if (@available(macOS 15.0, *)) { [(id<MTLResidencySet>)g_resset_obj commit]; }
|
||||
g_resset_dirty = false;
|
||||
}
|
||||
// Harness visibility for the flush cost, which sits OUTSIDE the moe_times setup/gpu
|
||||
// breakdown (timed around resset_flush in moe_submit, before ts_start). Returns whether
|
||||
// the set is active so colibri.c prints the METAL-RESSET line only when the gate is on --
|
||||
// stock output stays byte-identical.
|
||||
extern "C" int coli_metal_resset_stats(double *flush_s) {
|
||||
if (flush_s) *flush_s = g_t_resset_flush;
|
||||
return g_resset_enabled ? 1 : 0;
|
||||
}
|
||||
|
||||
// Persistent scratch buffers (grow-only) for the MoE pipeline.
|
||||
static id<MTLBuffer> g_gg, g_uu, g_hh, g_xg; static size_t g_gg_cap, g_uu_cap, g_hh_cap, g_xg_cap;
|
||||
static id<MTLBuffer> ensure(id<MTLBuffer> b, size_t *cap, size_t need) {
|
||||
@@ -284,6 +416,8 @@ extern "C" int coli_metal_init(void) {
|
||||
if (g_dev) return 1;
|
||||
if (getenv("COLI_METAL_UNTRACKED") && atoi(getenv("COLI_METAL_UNTRACKED")))
|
||||
g_res_opts = MTLResourceStorageModeShared | MTLResourceHazardTrackingModeUntracked;
|
||||
{ const char *e = getenv("COLI_RTOP8"); // default ON; COLI_RTOP8=0 opts out
|
||||
if (e && atoi(e) == 0) g_rtop8_par = 0; }
|
||||
@autoreleasepool {
|
||||
g_dev = MTLCreateSystemDefaultDevice();
|
||||
if (!g_dev) return 0;
|
||||
@@ -299,11 +433,45 @@ extern "C" int coli_metal_init(void) {
|
||||
auto P=[&](const char*n){ return [g_dev newComputePipelineStateWithFunction:[lib newFunctionWithName:@(n)] error:&err]; };
|
||||
g_a_rms=P("a_rmsnorm"); g_a_rope=P("a_rope"); g_a_copy=P("a_copy");
|
||||
g_a_qabs=P("a_qabs"); g_a_score=P("a_score"); g_a_smax=P("a_smax"); g_a_clat=P("a_clat"); g_a_ctx=P("a_ctx");
|
||||
g_a_add=P("a_add"); g_r_router=P("r_router"); g_r_top8=P("r_top8");
|
||||
if(!g_a_add||!g_r_router||!g_r_top8){ fprintf(stderr,"[metal] tail pipelines failed\n"); g_dev=nil; return 0; }
|
||||
g_a_add=P("a_add"); g_r_router=P("r_router"); g_r_top8=P("r_top8"); g_r_top8p=P("r_top8_par");
|
||||
if(!g_a_add||!g_r_router||!g_r_top8||!g_r_top8p){ fprintf(stderr,"[metal] tail pipelines failed\n"); g_dev=nil; return 0; }
|
||||
// r_top8_par's reduction hardcodes SIMD width 32 (shuffle-down offsets 16..1, one
|
||||
// 32-thread threadgroup per row). True on all Apple Silicon shipped to date, but a
|
||||
// non-32-width device would reduce wrongly AND race multiple lane-0 writers, so this
|
||||
// is a hard safety fact (g_rtop8_width_ok), not just a policy default: it gates BOTH
|
||||
// the engine dispatch site and the standalone coli_metal_rtop8 runner (degrade-to-safe,
|
||||
// same pattern as the pool/ring fallbacks elsewhere) — no caller can opt back into an
|
||||
// unsafe reduction on such a device, even by explicitly requesting par=1.
|
||||
if ([g_r_top8p threadExecutionWidth] != 32) {
|
||||
g_rtop8_width_ok = 0;
|
||||
if (g_rtop8_par)
|
||||
fprintf(stderr, "[metal] COLI_RTOP8 parallel top-8 disabled: threadExecutionWidth=%lu "
|
||||
"!= 32 (r_top8_par's reduction assumes 32-lane simdgroups) — serial "
|
||||
"r_top8 in use\n", (unsigned long)[g_r_top8p threadExecutionWidth]);
|
||||
g_rtop8_par = 0;
|
||||
}
|
||||
if (!g_gemv || !g_moe_gemv || !g_moe_silu || !g_a_rms || !g_a_rope || !g_a_copy ||
|
||||
!g_a_qabs || !g_a_score || !g_a_smax || !g_a_clat || !g_a_ctx) {
|
||||
fprintf(stderr, "[metal] pipeline failed\n"); g_dev = nil; return 0; }
|
||||
// E5 experiment: COLI_METAL_RESSET=1 -- see g_resset_obj comment above.
|
||||
if (getenv("COLI_METAL_RESSET") && atoi(getenv("COLI_METAL_RESSET"))) {
|
||||
if (@available(macOS 15.0, *)) {
|
||||
MTLResidencySetDescriptor *rd = [MTLResidencySetDescriptor new];
|
||||
rd.initialCapacity = 4096; // hint only (internal array presize), not a hard limit
|
||||
NSError *rerr = nil;
|
||||
id<MTLResidencySet> rs = [g_dev newResidencySetWithDescriptor:rd error:&rerr];
|
||||
if (rs) {
|
||||
[g_queue addResidencySet:rs];
|
||||
g_resset_obj = rs; g_resset_enabled = true;
|
||||
fprintf(stderr, "[METAL] residency-set: on (macOS 15+, moe_submit skips per-buffer useResource:)\n");
|
||||
} else {
|
||||
fprintf(stderr, "[METAL] residency-set create failed: %s -- stock per-CB residency path\n",
|
||||
rerr ? [[rerr localizedDescription] UTF8String] : "?");
|
||||
}
|
||||
} else {
|
||||
fprintf(stderr, "[METAL] COLI_METAL_RESSET=1 requested but OS < macOS 15 -- stock per-CB residency path\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
@@ -313,13 +481,29 @@ extern "C" void coli_metal_register(void *base, size_t len) {
|
||||
id<MTLBuffer> b = [g_dev newBufferWithBytesNoCopy:base length:len
|
||||
options:g_res_opts deallocator:nil];
|
||||
if (!b) return;
|
||||
std::lock_guard<std::mutex> lk(g_slab_mtx); // called from parallel expert_load threads
|
||||
for (auto &s : g_slabs) if (s.base == base) { s.len = len; s.buf = b; return; }
|
||||
g_slabs.push_back({base, len, b});
|
||||
id<MTLBuffer> old = nil; // E5: replaced wrapper on re-register of a live base (defensive)
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(g_slab_mtx); // called from parallel expert_load threads
|
||||
bool found = false;
|
||||
for (auto &s : g_slabs) if (s.base == base) { old = s.buf; s.len = len; s.buf = b; found = true; break; }
|
||||
if (!found) g_slabs.push_back({base, len, b});
|
||||
}
|
||||
// E5, outside g_slab_mtx (no Metal call under the slab lock), before returning. Invariant
|
||||
// defended: set membership mirrors g_slabs exactly -- a re-register of a live base must
|
||||
// drop the replaced wrapper from the set (ARC releases our reference, but the set retains
|
||||
// it and keeps its pages resident forever) before adding the new one. No in-tree caller
|
||||
// re-registers a live base today; defensive.
|
||||
if (old && old != b) resset_remove(old);
|
||||
if (old != b) resset_add(b);
|
||||
}
|
||||
extern "C" void coli_metal_unregister(void *base) {
|
||||
std::lock_guard<std::mutex> lk(g_slab_mtx);
|
||||
for (size_t i=0;i<g_slabs.size();i++) if (g_slabs[i].base==base) { g_slabs[i].buf=nil; g_slabs.erase(g_slabs.begin()+i); return; }
|
||||
id<MTLBuffer> b = nil;
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(g_slab_mtx);
|
||||
for (size_t i=0;i<g_slabs.size();i++) if (g_slabs[i].base==base) {
|
||||
b = g_slabs[i].buf; g_slabs[i].buf=nil; g_slabs.erase(g_slabs.begin()+i); break; }
|
||||
}
|
||||
if (b) resset_remove(b); // E5: outside g_slab_mtx; commits before the caller frees base
|
||||
}
|
||||
// Resolve a host pointer inside a registered slab to (buffer, gpuAddress). Returns nil if unknown.
|
||||
static id<MTLBuffer> resolve(const void *p, uint64_t *addr) {
|
||||
@@ -357,7 +541,14 @@ extern "C" void coli_metal_spin_start(void) {
|
||||
}
|
||||
extern "C" void coli_metal_spin_stop(void) { g_spin_run.store(false); }
|
||||
|
||||
extern "C" void coli_metal_shutdown(void) { coli_metal_spin_stop(); g_gemv=nil; g_queue=nil; g_dev=nil; g_tensor_count=g_tensor_bytes=0; }
|
||||
extern "C" void coli_metal_shutdown(void) {
|
||||
coli_metal_spin_stop();
|
||||
if (g_resset_enabled) {
|
||||
if (@available(macOS 15.0, *)) { [g_queue removeResidencySet:(id<MTLResidencySet>)g_resset_obj]; }
|
||||
}
|
||||
g_resset_obj=nil; g_resset_enabled=false; g_resset_dirty=false;
|
||||
g_gemv=nil; g_queue=nil; g_dev=nil; g_tensor_count=g_tensor_bytes=0;
|
||||
}
|
||||
extern "C" int coli_metal_available(void) { return g_dev != nil; }
|
||||
extern "C" void coli_metal_stats(size_t *c, size_t *b) { if(c)*c=g_tensor_count; if(b)*b=g_tensor_bytes; }
|
||||
extern "C" int coli_metal_mem_info(size_t *used, size_t *total) {
|
||||
@@ -597,12 +788,22 @@ extern "C" int coli_metal_layer_decode(float *x,
|
||||
// 5) silu(gate)*up + exact top-K select
|
||||
[e setComputePipelineState:g_moe_silu]; [e setBuffer:ash1_ offset:0 atIndex:0]; [e setBuffer:ash2_ offset:0 atIndex:1];
|
||||
[e dispatchThreads:MTLSizeMake((size_t)S*SI,1,1) threadsPerThreadgroup:MTLSizeMake(256,1,1)];
|
||||
{ [e setComputePipelineState:g_r_top8];
|
||||
{ // COLI_RTOP8 (default ON) swaps the serial 1-thread-per-row select for the exact-
|
||||
// match 1-simdgroup-per-row replica (same buffers/args; only pipeline+grid change).
|
||||
// E<=256 is required by r_top8_par's ch[8]/32-lane blocking contract; this call
|
||||
// site's E is always 256 today (layer_forward_rows' own architecture-shape gate in
|
||||
// colibri.c requires c->n_experts==256 to reach coli_metal_layer_decode at all —
|
||||
// see PR body "Scope statement") but the check is kept here too, defense-in-depth,
|
||||
// so a future relaxation of that gate (e.g. to admit REAP-pruned E=168 models into
|
||||
// the fused path) degrades safely to the serial kernel instead of mis-dispatching.
|
||||
int use_par = g_rtop8_par && g_rtop8_width_ok && E<=256;
|
||||
[e setComputePipelineState:use_par?g_r_top8p:g_r_top8];
|
||||
[e setBuffer:asig_ offset:0 atIndex:0]; [e setBuffer:rbB offset:rboff atIndex:1];
|
||||
[e setBuffer:aidx_ offset:0 atIndex:2]; [e setBuffer:aw_ offset:0 atIndex:3]; [e setBuffer:akeff_ offset:0 atIndex:4];
|
||||
[e setBytes:&E length:4 atIndex:5]; [e setBytes:&K length:4 atIndex:6]; [e setBytes:&Ksel length:4 atIndex:7];
|
||||
[e setBytes:&topp length:4 atIndex:8]; [e setBytes:&normk length:4 atIndex:9]; [e setBytes:&rscale length:4 atIndex:10];
|
||||
[e dispatchThreads:MTLSizeMake(S,1,1) threadsPerThreadgroup:MTLSizeMake(S,1,1)]; }
|
||||
if(use_par) [e dispatchThreadgroups:MTLSizeMake(S,1,1) threadsPerThreadgroup:MTLSizeMake(32,1,1)];
|
||||
else [e dispatchThreads:MTLSizeMake(S,1,1) threadsPerThreadgroup:MTLSizeMake(S,1,1)]; }
|
||||
BAR();
|
||||
// 6) shared down
|
||||
bind_gemv(e,shd_w,shd_s,shd_fmt,SI,AH,ash1_,ashout_,S);
|
||||
@@ -651,6 +852,44 @@ extern "C" int coli_metal_gemm(float *y, const float *x, const void *wp, const f
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Standalone single-kernel runner for the top-8 select (see backend_metal.h). Fresh
|
||||
// shared buffers per call (a test/probe path, not a hot path); grids exactly as the
|
||||
// engine dispatch site: serial = S threads of one S-wide threadgroup, parallel = S
|
||||
// threadgroups x 32 (one simdgroup per row). "par" is a REQUEST, not a guarantee: same
|
||||
// E<=256 and SIMD-width-32 host-side checks as the engine dispatch site gate the actual
|
||||
// pipeline choice, so a caller (including metal-test itself) can never reach the parallel
|
||||
// kernel out of contract by asking for it — par=1 with E>256, or on a non-32-wide device,
|
||||
// transparently runs the serial kernel instead and still returns 1 (success).
|
||||
extern "C" int coli_metal_rtop8(int par, const float *sig, const float *bias, int S, int E, int K,
|
||||
int Ksel, float topp, int normk, float rscale,
|
||||
int *idx, float *w, int *keff) {
|
||||
if (!g_dev || S < 1 || E < 1 || K < 1 || Ksel < 1 || Ksel > K) return 0;
|
||||
int use_par = par && g_r_top8p && g_rtop8_width_ok && E<=256;
|
||||
@autoreleasepool {
|
||||
id<MTLBuffer> bs=[g_dev newBufferWithBytes:sig length:(size_t)S*E*4 options:MTLResourceStorageModeShared];
|
||||
id<MTLBuffer> bb=[g_dev newBufferWithBytes:bias length:(size_t)E*4 options:MTLResourceStorageModeShared];
|
||||
id<MTLBuffer> bi=[g_dev newBufferWithLength:(size_t)S*K*4 options:MTLResourceStorageModeShared];
|
||||
id<MTLBuffer> bw=[g_dev newBufferWithLength:(size_t)S*K*4 options:MTLResourceStorageModeShared];
|
||||
id<MTLBuffer> bk=[g_dev newBufferWithLength:(size_t)S*4 options:MTLResourceStorageModeShared];
|
||||
if(!bs||!bb||!bi||!bw||!bk) return 0;
|
||||
memset(bi.contents,0xFF,(size_t)S*K*4); // poison: untouched slots stay visible
|
||||
id<MTLCommandBuffer> cb=[g_queue commandBuffer]; id<MTLComputeCommandEncoder> e=[cb computeCommandEncoder];
|
||||
[e setComputePipelineState:use_par?g_r_top8p:g_r_top8];
|
||||
[e setBuffer:bs offset:0 atIndex:0]; [e setBuffer:bb offset:0 atIndex:1];
|
||||
[e setBuffer:bi offset:0 atIndex:2]; [e setBuffer:bw offset:0 atIndex:3]; [e setBuffer:bk offset:0 atIndex:4];
|
||||
[e setBytes:&E length:4 atIndex:5]; [e setBytes:&K length:4 atIndex:6]; [e setBytes:&Ksel length:4 atIndex:7];
|
||||
[e setBytes:&topp length:4 atIndex:8]; [e setBytes:&normk length:4 atIndex:9]; [e setBytes:&rscale length:4 atIndex:10];
|
||||
if(use_par) [e dispatchThreadgroups:MTLSizeMake((NSUInteger)S,1,1) threadsPerThreadgroup:MTLSizeMake(32,1,1)];
|
||||
else [e dispatchThreads:MTLSizeMake((NSUInteger)S,1,1) threadsPerThreadgroup:MTLSizeMake((NSUInteger)S,1,1)];
|
||||
[e endEncoding]; [cb commit]; [cb waitUntilCompleted];
|
||||
if(cb.status==MTLCommandBufferStatusError){ fprintf(stderr,"[metal] rtop8 cmdbuf error\n"); return 0; }
|
||||
memcpy(idx,bi.contents,(size_t)S*K*4);
|
||||
memcpy(w,bw.contents,(size_t)S*K*4);
|
||||
memcpy(keff,bk.contents,(size_t)S*4);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
extern "C" void coli_metal_tensor_free(ColiMetalTensor *t) {
|
||||
if (!t) return;
|
||||
g_tensor_count--; g_tensor_bytes -= t->wbytes;
|
||||
@@ -668,6 +907,9 @@ static id<MTLCommandBuffer> moe_submit(int nb, int D, int Iinter, int fmt,
|
||||
const float *xg, const int *xoff, const int *nr, int R,
|
||||
id<MTLBuffer> xg_buf, id<MTLBuffer> gg_buf, id<MTLBuffer> uu_buf, id<MTLBuffer> hh_buf) {
|
||||
if (!g_dev || (fmt != 1 && fmt != 2)) return nil;
|
||||
if (g_resset_enabled) { // E5: commit any pending slab adds before we may skip useResource:
|
||||
double t0 = mnow(); resset_flush(); g_t_resset_flush += mnow() - t0; // METAL-RESSET line
|
||||
}
|
||||
double ts_start = mnow();
|
||||
std::vector<uint64_t> ag(nb),au(nb),ad(nb),sgv(nb),suv(nb),sdv(nb);
|
||||
std::vector<id<MTLBuffer>> use; use.reserve(nb*2);
|
||||
@@ -689,7 +931,18 @@ static id<MTLCommandBuffer> moe_submit(int nb, int D, int Iinter, int fmt,
|
||||
memcpy([xg_buf contents], xg, (size_t)R*D*4);
|
||||
|
||||
id<MTLCommandBuffer> cb=[g_queue commandBuffer]; id<MTLComputeCommandEncoder> e=[cb computeCommandEncoder];
|
||||
for(auto&b:use) [e useResource:b usage:MTLResourceUsageRead];
|
||||
// E5 (COLI_METAL_RESSET=1): the queue-attached MTLResidencySet already guarantees these
|
||||
// buffers are resident, so skip the per-buffer declaration whose count scales with LRU
|
||||
// cache size (mechanism history v5). Residency sets don't do hazard tracking (Apple docs),
|
||||
// but none was load-bearing here: every buffer in `use` is MTLResourceUsageRead-only and
|
||||
// referenced only indirectly (moe_gemv dereferences waddr[]/saddr[] baked into bag/bsg's
|
||||
// contents), so there's no GPU-side write to serialize against; the one real hazard -- a
|
||||
// slab unregistered+freed+reused while an async in-flight CB still reads it -- is a
|
||||
// CPU-write race outside Metal's hazard tracking either way, held by the engine's own slot
|
||||
// lifecycle, not by useResource:. See SUMMARY.md UNCERTAINTIES.
|
||||
if (!g_resset_enabled) {
|
||||
for(auto&b:use) [e useResource:b usage:MTLResourceUsageRead];
|
||||
}
|
||||
auto gemv=[&](id<MTLBuffer> wa,id<MTLBuffer> sa,id<MTLBuffer> xin,id<MTLBuffer> y,int O,int K,int Kin){
|
||||
int NT=R*O;
|
||||
[e setComputePipelineState:g_moe_gemv];
|
||||
|
||||
@@ -15,6 +15,9 @@ Run GLM-5.2 (744B) locally on CPU with roughly 15-26 GB of RAM.
|
||||
|
||||
Configuration through environment variables or flags (also valid after the subcommand):
|
||||
COLI_MODEL=<dir> model directory (default /home/vincenzo/glm52_i4)
|
||||
COLI_MODEL_MIRROR=<dir> second copy of the model on another drive: expert reads
|
||||
are split across both SSDs (COLI_DISK_WEIGHTS=9,3 sets the
|
||||
primary,mirror bandwidth ratio; default: measured at startup)
|
||||
--ram N RAM budget in GB (automatically sizes the expert cache)
|
||||
--repin N adapt RAM/VRAM experts every N tokens
|
||||
--topp P adaptive expert top-p --topk N fixed top-k
|
||||
@@ -54,16 +57,20 @@ from version import __version__ as _version
|
||||
# guess is right (e.g. a custom packaging layout).
|
||||
_EXE = ".exe" if sys.platform == "win32" else ""
|
||||
_LIBEXEC = os.path.join(os.path.dirname(HERE), "libexec", "colibri")
|
||||
_here_colibri = os.path.join(HERE, "colibri" + _EXE)
|
||||
_here_glm = os.path.join(HERE, "glm" + _EXE)
|
||||
|
||||
if os.environ.get("COLI_ENGINE"):
|
||||
GLM = os.environ["COLI_ENGINE"]
|
||||
TOOLS = os.path.join(os.path.dirname(GLM), "tools")
|
||||
elif os.path.exists(_here_colibri):
|
||||
GLM = _here_colibri
|
||||
TOOLS = os.path.join(HERE, "tools")
|
||||
elif os.path.exists(_here_glm):
|
||||
GLM = _here_glm
|
||||
TOOLS = os.path.join(HERE, "tools")
|
||||
else:
|
||||
GLM = os.path.join(_LIBEXEC, "glm" + _EXE)
|
||||
GLM = os.path.join(_LIBEXEC, "colibri" + _EXE)
|
||||
TOOLS = os.path.join(_LIBEXEC, "tools")
|
||||
sys.path.insert(0, _LIBEXEC) # so `import resource_plan`, `doctor`, `openai_server` still resolve
|
||||
|
||||
@@ -233,6 +240,55 @@ def env_for(a):
|
||||
gpu=f" · VRAM {format_bytes(vt['budget_bytes'])}" if has_cuda and vt["devices"] else " · CPU"
|
||||
print(f" {C.dim}[PLAN] RAM {format_bytes(rt['budget_bytes'])} · cap {rt['cache_slots_per_layer']}/layer{gpu}{C.r}",file=sys.stderr)
|
||||
else:
|
||||
# Windows: a bare `coli chat` (no --gpu/--vram/--auto-tier) used to ALWAYS
|
||||
# run CPU-only, even on a CUDA build with a GPU present — cuda_binary()
|
||||
# returned False on Windows (see above), and nothing set COLI_CUDA without
|
||||
# an explicit flag. Now that detection works, auto-enable the GPU when one
|
||||
# is detected so `coli chat` Just Works. Scoped to Windows: Linux already
|
||||
# has working detection + the explicit-flag UX, and changing bare-chat
|
||||
# semantics there is out of scope. Falls back to CPU with a warning if
|
||||
# nvidia-smi is missing (discover_gpus can't size VRAM without it).
|
||||
# An explicit COLI_CUDA=0 in the environment must win over the implicit
|
||||
# auto-enable: before this check, a Windows user setting COLI_CUDA=0 for
|
||||
# a CPU baseline silently got a ~12.6 GB VRAM expert tier anyway (the
|
||||
# engine's "CPU" rows were GPU-assisted). --gpu none remains the
|
||||
# canonical hard off-switch (works on every platform, also clears the
|
||||
# CUDA_* sizing vars).
|
||||
if (sys.platform == "win32" and a.gpu is None and not a.vram
|
||||
and e.get("COLI_CUDA") != "0"):
|
||||
if cuda_binary():
|
||||
from resource_plan import discover_gpus, build_plan, environment_for_plan, format_bytes
|
||||
gpus = discover_gpus()
|
||||
if gpus:
|
||||
e["COLI_CUDA"]="1"
|
||||
e.setdefault("COLI_GPUS", ",".join(str(g["index"]) for g in gpus))
|
||||
# Reuse the planner so the expert-tier VRAM budget is the real
|
||||
# free VRAM minus the 2 GB reserve — not a guess. Same machinery
|
||||
# as --auto-tier, just without requiring the user to pass it.
|
||||
ram,ctx,devices,vram_req = resource_request(a, e)
|
||||
try:
|
||||
plan=build_plan(a.model,ram,ctx,devices,vram_req,policy=a.policy)
|
||||
e.update(environment_for_plan(plan,e,cuda_enabled=True))
|
||||
vt=plan["tiers"]["vram"]
|
||||
names=",".join(g["name"].strip() for g in gpus)
|
||||
print(f" {C.dim}[GPU] auto-enabled CUDA · {names} · "
|
||||
f"{format_bytes(vt['budget_bytes'])} expert tier{C.r}", file=sys.stderr)
|
||||
except (OSError,ValueError,json.JSONDecodeError) as error:
|
||||
# Plan failed (e.g. model dir unreadable): don't block the
|
||||
# run, just leave the unsized COLI_CUDA=1 and let the engine
|
||||
# pick its own budget. Engine handles a missing budget.
|
||||
print(f" {C.yel}[GPU] auto-enable: could not size VRAM ({error}); "
|
||||
f"using engine default{C.r}", file=sys.stderr)
|
||||
else:
|
||||
print(f" {C.yel}[GPU] coli_cuda.dll present but nvidia-smi not found on PATH "
|
||||
f"(cannot size VRAM); running CPU-only. Add nvidia-smi to PATH or pass "
|
||||
f"--vram N to enable CUDA.{C.r}", file=sys.stderr)
|
||||
# else: CPU build (no coli_cuda.dll) — stay silent, CPU is correct.
|
||||
elif e.get("COLI_CUDA") == "0":
|
||||
# honoured off-switch: also drop stale device/sizing vars so the
|
||||
# engine can't be re-enabled by leftovers (same as --gpu none).
|
||||
e.pop("COLI_GPU",None); e.pop("COLI_GPUS",None)
|
||||
e.pop("CUDA_EXPERT_GB",None); e.pop("CUDA_DENSE",None)
|
||||
# --gpu/--vram SENZA --auto-tier: prima venivano ignorati in silenzio e il run
|
||||
# partiva CPU-only senza alcun avviso — benchmark "GPU" pubblicati per errore (#121).
|
||||
if a.gpu is not None:
|
||||
@@ -241,13 +297,13 @@ def env_for(a):
|
||||
e["COLI_CUDA"]="0"; e.pop("CUDA_EXPERT_GB",None); e.pop("CUDA_DENSE",None)
|
||||
else:
|
||||
if not cuda_binary():
|
||||
sys.exit(f"{C.yel}--gpu needs the CUDA build:{C.r} make glm CUDA=1 (this binary is CPU-only)")
|
||||
sys.exit(f"{C.yel}--gpu needs the CUDA build:{C.r} make colibri CUDA=1 (this binary is CPU-only)")
|
||||
e["COLI_CUDA"]="1"
|
||||
if a.gpu!="auto": e["COLI_GPUS"]=a.gpu
|
||||
e.setdefault("CUDA_DENSE","1")
|
||||
if a.vram and a.gpu!="none":
|
||||
if not cuda_binary():
|
||||
sys.exit(f"{C.yel}--vram needs the CUDA build:{C.r} make glm CUDA=1 (this binary is CPU-only)")
|
||||
sys.exit(f"{C.yel}--vram needs the CUDA build:{C.r} make colibri CUDA=1 (this binary is CPU-only)")
|
||||
e["COLI_CUDA"]="1"; e["CUDA_EXPERT_GB"]=str(a.vram)
|
||||
return e
|
||||
|
||||
@@ -421,8 +477,8 @@ def cmd_build(a):
|
||||
banner("build")
|
||||
if not os.path.exists(os.path.join(HERE, "Makefile")):
|
||||
sys.exit(f"{C.yel}coli build{C.r} only works from a source checkout (this is an installed copy).\n"
|
||||
f" Clone https://github.com/JustVugg/colibri and run ./setup.sh, or make -C c glm.")
|
||||
sys.exit(subprocess.call(["make","-C",HERE,"glm"]))
|
||||
f" Clone https://github.com/JustVugg/colibri and run ./setup.sh, or make -C c colibri.")
|
||||
sys.exit(subprocess.call(["make","-C",HERE,"colibri"]))
|
||||
|
||||
def cmd_info(a):
|
||||
banner("info")
|
||||
@@ -802,7 +858,7 @@ def cmd_stop(a):
|
||||
if "coli" in cmd and " serve" in cmd and pid!=os.getpid():
|
||||
if not any(p==pid for p,_ in targets): targets.append((pid,"coli serve (cmdline)"))
|
||||
comm=open(f"/proc/{pd}/comm").read().strip()
|
||||
if comm in ("glm","exe","olmoe"):
|
||||
if comm in ("colibri","glm","exe","olmoe"):
|
||||
env=open(f"/proc/{pd}/environ","rb").read().replace(b"\0",b"\n").decode("utf-8","replace")
|
||||
if "SERVE=1" in env: targets.append((pid,f"engine `{comm}` (SERVE=1)"))
|
||||
except (OSError,PermissionError): continue
|
||||
|
||||
+1168
-1352
File diff suppressed because it is too large
Load Diff
+17
-7
@@ -13,22 +13,32 @@ static inline float *coli_kv_row(float *base, int position, int width)
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
unsigned long long id, bytes;
|
||||
unsigned long long id, bytes, gbytes;
|
||||
int slot, max_tokens;
|
||||
float temperature, top_p;
|
||||
} ColiSubmit;
|
||||
|
||||
/* Parse the textual header. The payload is read separately using `bytes`, so
|
||||
* it may contain newlines. Reject trailing fields to keep framing unambiguous. */
|
||||
* it may contain newlines. Reject trailing fields to keep framing unambiguous.
|
||||
* Optional 7th field `gbytes`: length of a per-request grammar (raw GBNF, or a
|
||||
* JSON-Schema compiled engine-side) appended to the payload AFTER the prompt
|
||||
* bytes. 6-field headers remain valid (gbytes = 0). */
|
||||
static inline int coli_submit_parse(const char *line, ColiSubmit *s)
|
||||
{
|
||||
char tail;
|
||||
if (!line || !s ||
|
||||
sscanf(line, "SUBMIT %llu %d %llu %d %f %f %c", &s->id, &s->slot,
|
||||
if (!line || !s) return 0;
|
||||
s->gbytes = 0;
|
||||
if (sscanf(line, "SUBMIT %llu %d %llu %d %f %f %llu %c", &s->id, &s->slot,
|
||||
&s->bytes, &s->max_tokens, &s->temperature, &s->top_p,
|
||||
&tail) != 6)
|
||||
return 0;
|
||||
return s->id > 0 && s->bytes <= (16u << 20) && s->slot >= 0 && s->max_tokens >= 1 &&
|
||||
&s->gbytes, &tail) != 7) {
|
||||
s->gbytes = 0;
|
||||
if (sscanf(line, "SUBMIT %llu %d %llu %d %f %f %c", &s->id, &s->slot,
|
||||
&s->bytes, &s->max_tokens, &s->temperature, &s->top_p,
|
||||
&tail) != 6)
|
||||
return 0;
|
||||
}
|
||||
return s->id > 0 && s->bytes <= (16u << 20) && s->gbytes <= (1u << 20) &&
|
||||
s->slot >= 0 && s->max_tokens >= 1 &&
|
||||
isfinite(s->temperature) && isfinite(s->top_p) &&
|
||||
s->temperature >= 0 && s->temperature <= 2 &&
|
||||
s->top_p > 0 && s->top_p <= 1;
|
||||
|
||||
+121
@@ -0,0 +1,121 @@
|
||||
/* kv_persist.h — .coli_kv on-disk KV cache persistence.
|
||||
* Conversations reopen warm across engine restarts: the compressed MLA KV-cache
|
||||
* is appended incrementally after every turn, crash-safe (nrec written last).
|
||||
* Include after Model/KVState/Cfg are defined; requires now_s() and g_draft. */
|
||||
#ifndef KV_PERSIST_H
|
||||
#define KV_PERSIST_H
|
||||
|
||||
static int g_kvsave=1;
|
||||
#define KV_MAGIC "COLIKV1\0"
|
||||
|
||||
static void kv_hdr(Model *m, int32_t *h, int nrec){
|
||||
Cfg *c=&m->c; int nic=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->Ic && m->Ic[i]) nic++;
|
||||
h[0]=c->n_layers; h[1]=c->kv_lora; h[2]=c->qk_rope;
|
||||
h[3]=m->has_dsa?c->index_hd:0; h[4]=nic; h[5]=c->vocab; h[6]=nrec; h[7]=0;
|
||||
}
|
||||
|
||||
static int64_t kv_rec_bytes(Model *m){
|
||||
Cfg *c=&m->c;
|
||||
int64_t rec = 4 + (int64_t)c->n_layers*(c->kv_lora+c->qk_rope)*4;
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i]) rec+=(int64_t)c->index_hd*4;
|
||||
return rec;
|
||||
}
|
||||
|
||||
static int kv_disk_open(Model *m){
|
||||
KVState *k=m->kv;
|
||||
if(k->disk_fp) return 1;
|
||||
k->disk_fp=fopen(k->disk_path,"r+b");
|
||||
if(!k->disk_fp){
|
||||
k->disk_fp=fopen(k->disk_path,"wb");
|
||||
if(!k->disk_fp) return 0;
|
||||
int32_t h[8]; kv_hdr(m,h,0);
|
||||
fwrite(KV_MAGIC,1,8,k->disk_fp); fwrite(h,4,8,k->disk_fp);
|
||||
fflush(k->disk_fp);
|
||||
fclose(k->disk_fp);
|
||||
k->disk_fp=fopen(k->disk_path,"r+b");
|
||||
if(!k->disk_fp) return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static void kv_disk_truncate(Model *m, int nrec){
|
||||
if(!g_kvsave) return;
|
||||
KVState *k=m->kv;
|
||||
if(k->disk_fp){ fclose(k->disk_fp); k->disk_fp=NULL; }
|
||||
FILE *f=fopen(k->disk_path,"r+b");
|
||||
if(!f){ k->disk_nrec=0; return; }
|
||||
k->disk_nrec=nrec;
|
||||
int32_t nr=nrec; fseek(f,8+6*4,SEEK_SET); fwrite(&nr,4,1,f);
|
||||
fflush(f); fclose(f);
|
||||
}
|
||||
|
||||
static void kv_disk_reset(Model *m){ kv_disk_truncate(m,0); }
|
||||
|
||||
static void kv_disk_append(Model *m, const int *hist, int len){
|
||||
KVState *k=m->kv;
|
||||
if(!g_kvsave || len<=k->disk_nrec) return;
|
||||
Cfg *c=&m->c;
|
||||
if(!kv_disk_open(m)) return;
|
||||
FILE *f=k->disk_fp;
|
||||
int64_t rec = kv_rec_bytes(m);
|
||||
if(rec > k->disk_buf_cap){
|
||||
uint8_t *nb=realloc(k->disk_buf, rec);
|
||||
if(!nb) return;
|
||||
k->disk_buf=nb; k->disk_buf_cap=rec;
|
||||
}
|
||||
fseek(f, 8+8*4 + (int64_t)k->disk_nrec*rec, SEEK_SET);
|
||||
for(int p=k->disk_nrec;p<len;p++){
|
||||
uint8_t *b=k->disk_buf;
|
||||
*(int32_t*)b = hist[p]; b+=4;
|
||||
for(int i=0;i<c->n_layers;i++){
|
||||
memcpy(b, m->Lc[i]+(int64_t)p*c->kv_lora, (size_t)c->kv_lora*4); b+=c->kv_lora*4;
|
||||
memcpy(b, m->Rc[i]+(int64_t)p*c->qk_rope,(size_t)c->qk_rope*4); b+=c->qk_rope*4;
|
||||
}
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i]){
|
||||
memcpy(b, m->Ic[i]+(int64_t)p*c->index_hd, (size_t)c->index_hd*4); b+=c->index_hd*4;
|
||||
}
|
||||
fwrite(k->disk_buf, 1, (size_t)rec, f);
|
||||
}
|
||||
fflush(f);
|
||||
int32_t nr=len; fseek(f,8+6*4,SEEK_SET); fwrite(&nr,4,1,f);
|
||||
fflush(f);
|
||||
k->disk_nrec=len;
|
||||
}
|
||||
|
||||
static int kv_disk_load(Model *m, int *hist, int maxctx){
|
||||
if(!g_kvsave) return 0;
|
||||
KVState *k=m->kv;
|
||||
Cfg *c=&m->c;
|
||||
FILE *f=fopen(k->disk_path,"rb"); if(!f) return 0;
|
||||
char mg[8]; int32_t h[8], w[8]; kv_hdr(m,w,0);
|
||||
if(fread(mg,1,8,f)!=8 || memcmp(mg,KV_MAGIC,8) || fread(h,4,8,f)!=8 ||
|
||||
h[0]!=w[0]||h[1]!=w[1]||h[2]!=w[2]||h[3]!=w[3]||h[4]!=w[4]||h[5]!=w[5]){
|
||||
fprintf(stderr,"[KV] ignoring .coli_kv from a different model or version\n"); fclose(f); return 0; }
|
||||
int nrec=h[6];
|
||||
if(nrec<1){ fclose(f); return 0; }
|
||||
if(nrec>=maxctx-8-g_draft){
|
||||
fprintf(stderr,"[KV] saved conversation (%d tokens) exceeds the context: starting over\n",nrec);
|
||||
fclose(f); return 0; }
|
||||
double t0=now_s();
|
||||
for(int p=0;p<nrec;p++){
|
||||
int32_t tk; if(fread(&tk,4,1,f)!=1){ nrec=p; break; } hist[p]=tk;
|
||||
for(int i=0;i<c->n_layers;i++){
|
||||
if(fread(m->Lc[i]+(int64_t)p*c->kv_lora, 4, c->kv_lora, f)!=(size_t)c->kv_lora ||
|
||||
fread(m->Rc[i]+(int64_t)p*c->qk_rope, 4, c->qk_rope, f)!=(size_t)c->qk_rope){ nrec=p; goto out; }
|
||||
}
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i])
|
||||
if(fread(m->Ic[i]+(int64_t)p*c->index_hd, 4, c->index_hd, f)!=(size_t)c->index_hd){ nrec=p; goto out; }
|
||||
}
|
||||
out:
|
||||
fclose(f);
|
||||
if(nrec>0){
|
||||
if(m->has_mtp) m->kv_start[c->n_layers]=-1;
|
||||
fprintf(stderr,"[KV] resumed conversation from disk: %d tokens in %.1fs (no re-prefill)\n",
|
||||
nrec, now_s()-t0);
|
||||
}
|
||||
k->disk_nrec=nrec;
|
||||
return nrec;
|
||||
}
|
||||
|
||||
#endif /* KV_PERSIST_H */
|
||||
@@ -5,6 +5,15 @@
|
||||
* Densa (embed, attn, router, norme, lm_head) residente in RAM (float32).
|
||||
* Expert letti dal disco on-demand via pread+fadvise(DONTNEED), cache LRU per-layer.
|
||||
* Matmul multi-thread con OpenMP (niente BLAS).
|
||||
*
|
||||
* ENV VARS:
|
||||
* PILOT=0/1/2/3 : 0=no prefetch, 1=1-layer lookahead, 2=2-layer, 3=3-layer lookahead
|
||||
* HOT=N : pin top-N hot experts per layer permanently (never evict)
|
||||
* WARMUP=N : tokens before hot pinning activates (default 5)
|
||||
* WIDE=N : prefetch top-K*N candidates (default 1, try 2 or 3)
|
||||
* SMOOTH=F : EMA coefficient for routing momentum (default 0.3, range 0.0-0.95)
|
||||
* CONF_LIMIT=F : cumulative gate probability threshold for prefetch cutoff (default 0.92)
|
||||
* (expert queue is sorted by eid for SSD read locality)
|
||||
*/
|
||||
#define _GNU_SOURCE
|
||||
#include <stdio.h>
|
||||
@@ -12,11 +21,22 @@
|
||||
#include <string.h>
|
||||
#include <math.h>
|
||||
#include <time.h>
|
||||
#include <pthread.h>
|
||||
#if defined(__APPLE__) || defined(__linux__) || defined(__FreeBSD__)
|
||||
#include <sys/resource.h>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include "st.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#define sleep_ms(ms) Sleep(ms)
|
||||
#else
|
||||
#define sleep_ms(ms) usleep((ms) * 1000)
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
/* ---------- config ---------- */
|
||||
typedef struct {
|
||||
int hidden, n_layers, n_heads, n_kv_heads, head_dim;
|
||||
@@ -33,22 +53,61 @@ typedef struct {
|
||||
* Ogni weight [out,in] tenuto come int8 (per-riga) + scala float per riga.
|
||||
* Cosi' la RAM-cache scende da 4 byte/param (f32) a 1 byte/param: e' il
|
||||
* meccanismo che fa stare GLM-5.2 nei 15 GB. dequant-on-use nel matmul. */
|
||||
typedef struct { int eid; int8_t *g, *u, *d; float *gs, *us, *ds; uint64_t used; } Slot;
|
||||
/* pinned=1 means this slot is strongly preferred to keep (hot expert); it will
|
||||
* not be evicted during normal LRU eviction, but may be displaced under extreme
|
||||
* cache pressure when all slots are pinned or in-flight. */
|
||||
typedef struct { int eid; int pinned; int8_t *g, *u, *d; float *gs, *us, *ds; uint64_t used; } Slot;
|
||||
typedef struct { Slot *slots; int n, cap; } LCache;
|
||||
|
||||
typedef struct {
|
||||
Cfg c;
|
||||
shards S;
|
||||
int quant_bits; /* bit di quantizzazione degli expert (2..8); storage int8, niente f32 (#134) */
|
||||
int quant_bits;
|
||||
float *embed, *lm_head, *final_norm;
|
||||
Layer *L;
|
||||
LCache *cache; /* [n_layers] */
|
||||
uint64_t clock, hits, miss;
|
||||
/* kv-cache per-layer: K,V come [H * maxT * head_dim] */
|
||||
float **K, **V; int kv_len, max_t;
|
||||
double dense_load_s;
|
||||
/* IMPROVEMENT 2: expert frequency heatmap */
|
||||
uint32_t *freq;
|
||||
int freq_token_count, hot_pinned, hot_n, warmup_tokens;
|
||||
int token_count;
|
||||
/* PREDICTION IMPROVEMENT A: per-layer EMA of gate logits across tokens.
|
||||
* momentum_logits[l*E .. (l+1)*E-1] = EMA of gate outputs for layer l.
|
||||
* Used exclusively by the PILOT prefetcher to stabilise routing predictions
|
||||
* across tokens; does NOT affect actual MoE routing (pr is unchanged). */
|
||||
float *momentum_logits; /* [n_layers * n_experts], EMA of gate logits */
|
||||
float pilot_smooth; /* SMOOTH env: EMA coefficient 0.0-0.9 (default 0.3) */
|
||||
uint8_t *is_pinned; /* [n_layers * n_experts], 1 if expert is globally pinned */
|
||||
uint8_t *is_queued; /* [n_layers * n_experts], 1 if expert is currently in the prefetch queue */
|
||||
float pilot_conf_limit; /* CONF_LIMIT env: cumulative gate probability threshold (e.g. 0.92) */
|
||||
} Model;
|
||||
|
||||
static pthread_mutex_t g_pilot_mx = PTHREAD_MUTEX_INITIALIZER;
|
||||
static struct { int l, e; } pilot_q[4096];
|
||||
static volatile unsigned pilot_r = 0, pilot_w = 0;
|
||||
static Model *pilot_m = NULL;
|
||||
static int g_pilot = 0;
|
||||
static int g_wide = 1; /* IMPROVEMENT 4: top-K * g_wide candidates prefetched */
|
||||
|
||||
static void pilot_prefetch(Model *m, int lnext, const float *x, int S);
|
||||
static void *pilot_worker(void *arg);
|
||||
static void ensure_pilot_worker_started(Model *m);
|
||||
static void slot_ensure_allocated(Model *m, Slot *s);
|
||||
|
||||
static void ensure_pilot_worker_started(Model *m) {
|
||||
if (!pilot_m) {
|
||||
pilot_m = m;
|
||||
pthread_t t;
|
||||
if (pthread_create(&t, NULL, pilot_worker, NULL) != 0) {
|
||||
fprintf(stderr, "Error: Failed to create pilot prefetch worker thread\n");
|
||||
exit(1);
|
||||
}
|
||||
pthread_detach(t);
|
||||
}
|
||||
}
|
||||
|
||||
/* ---------- utility ---------- */
|
||||
static double now_s(void) { struct timespec t; clock_gettime(CLOCK_MONOTONIC, &t); return t.tv_sec + t.tv_nsec*1e-9; }
|
||||
#if defined(__APPLE__)
|
||||
@@ -210,51 +269,224 @@ static void model_init(Model *m, const char *snap, int cap, int bits) {
|
||||
#undef LD
|
||||
}
|
||||
m->cache = calloc(c->n_layers, sizeof(LCache));
|
||||
for (int i = 0; i < c->n_layers; i++) { m->cache[i].cap = cap; m->cache[i].slots = calloc(cap, sizeof(Slot)); }
|
||||
for (int i = 0; i < c->n_layers; i++) {
|
||||
m->cache[i].cap = cap;
|
||||
m->cache[i].slots = calloc(cap, sizeof(Slot));
|
||||
}
|
||||
/* IMPROVEMENT 2: frequency heatmap for hot expert pinning */
|
||||
m->freq = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint32_t));
|
||||
m->hot_pinned = 0; m->freq_token_count = 0;
|
||||
m->hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0;
|
||||
m->warmup_tokens = getenv("WARMUP") ? atoi(getenv("WARMUP")) : 5;
|
||||
m->token_count = 0;
|
||||
/* PREDICTION A: routing momentum — EMA of gate logits across tokens.
|
||||
* Initialized to zero; first token sets EMA = fresh logits. */
|
||||
m->momentum_logits = calloc((size_t)c->n_layers * c->n_experts, sizeof(float));
|
||||
float sv = getenv("SMOOTH") ? (float)atof(getenv("SMOOTH")) : 0.3f;
|
||||
if (sv < 0.f) sv = 0.f; if (sv > 0.95f) sv = 0.95f;
|
||||
m->pilot_smooth = sv;
|
||||
m->is_pinned = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint8_t));
|
||||
m->is_queued = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint8_t));
|
||||
float cl = getenv("CONF_LIMIT") ? (float)atof(getenv("CONF_LIMIT")) : 0.92f;
|
||||
if (cl < 0.1f) cl = 0.1f; if (cl > 1.0f) cl = 1.0f;
|
||||
m->pilot_conf_limit = cl;
|
||||
m->dense_load_s = now_s() - t0;
|
||||
|
||||
// Persistent Hot Pinning: try to load hot_pinned.bin
|
||||
char pinpath[512];
|
||||
snprintf(pinpath, sizeof(pinpath), "%s/hot_pinned.bin", snap);
|
||||
FILE *pinf = fopen(pinpath, "rb");
|
||||
if (pinf) {
|
||||
size_t expected_size = (size_t)c->n_layers * c->n_experts;
|
||||
if (fread(m->is_pinned, 1, expected_size, pinf) == expected_size) {
|
||||
m->hot_pinned = 1;
|
||||
printf("[HOT] Loaded persistent pinning from %s\n", pinpath);
|
||||
|
||||
if (g_pilot) {
|
||||
ensure_pilot_worker_started(m);
|
||||
for (int l = 0; l < c->n_layers; l++) {
|
||||
for (int e = 0; e < c->n_experts; e++) {
|
||||
if (m->is_pinned[l * c->n_experts + e]) {
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
if (w - r < 4096) {
|
||||
pilot_q[w & 4095].l = l; pilot_q[w & 4095].e = e;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
m->is_queued[l * c->n_experts + e] = 1;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
__atomic_store_n(&pilot_w, w + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
printf("[HOT] Pre-loading pinned experts into cache...\n");
|
||||
double t_wait = now_s();
|
||||
while (1) {
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_ACQUIRE);
|
||||
if (r == w) break;
|
||||
sleep_ms(2);
|
||||
}
|
||||
printf("[HOT] Pre-loaded in %.1fs!\n", now_s() - t_wait);
|
||||
}
|
||||
}
|
||||
fclose(pinf);
|
||||
}
|
||||
}
|
||||
|
||||
/* legge un weight dal disco (streaming) e lo quantizza in q[O,I]+scale[O].
|
||||
* Container pre-quantizzato (convert_olmoe.py: int8 + scale f32 in "name.qs"):
|
||||
* lettura raw diretta — meta' I/O e zero quantize_rows a runtime. Prima di
|
||||
* questa patch il container int8 causava SIGBUS (st_read_f32 su tensori I8). */
|
||||
static void load_expert_w(Model *m, const char *name, int8_t *q, float *scale, int O, int I, float *tmp) {
|
||||
st_tensor *t = st_find(&m->S, name);
|
||||
if (t && t->dtype == 3) { /* I8/U8: container colibri */
|
||||
char qs[300]; snprintf(qs, sizeof(qs), "%s.qs", name);
|
||||
st_read_raw(&m->S, name, q, 1);
|
||||
st_read_f32(&m->S, qs, scale, 1);
|
||||
return;
|
||||
static void slot_ensure_allocated(Model *m, Slot *s) {
|
||||
if (s->g) return;
|
||||
Cfg *c = &m->c;
|
||||
int64_t ng = (int64_t)c->inter * c->hidden;
|
||||
int64_t nd = (int64_t)c->hidden * c->inter;
|
||||
int8_t *w_block = malloc(ng + ng + nd);
|
||||
if (!w_block) {
|
||||
fprintf(stderr, "Error: Out of memory allocating slot weights block\n");
|
||||
exit(1);
|
||||
}
|
||||
st_read_f32(&m->S, name, tmp, 1); /* pread + fadvise DONTNEED */
|
||||
quantize_rows(tmp, q, scale, O, I, m->quant_bits);
|
||||
s->g = w_block;
|
||||
s->u = w_block + ng;
|
||||
s->d = w_block + ng + ng;
|
||||
float *s_block = falloc(c->inter + c->inter + c->hidden);
|
||||
s->gs = s_block;
|
||||
s->us = s_block + c->inter;
|
||||
s->ds = s_block + c->inter + c->inter;
|
||||
s->pinned = 0;
|
||||
}
|
||||
|
||||
static void load_expert_merged(Model *m, int layer, int eid, Slot *s) {
|
||||
char nm[256], qsnm[256];
|
||||
snprintf(nm, sizeof(nm), "model.layers.%d.mlp.experts.%d.merged_weight", layer, eid);
|
||||
snprintf(qsnm, sizeof(qsnm), "model.layers.%d.mlp.experts.%d.qs", layer, eid);
|
||||
st_read_raw(&m->S, nm, s->g, 1);
|
||||
st_read_f32(&m->S, qsnm, s->gs, 0); /* scales are F32; use typed reader for dtype safety */
|
||||
}
|
||||
|
||||
/* ---------- cache expert: ritorna i pesi quantizzati (q+scale) da cache o disco ---------- */
|
||||
static void expert_get(Model *m, int layer, int eid, Slot **out) {
|
||||
LCache *lc = &m->cache[layer];
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
for (int i = 0; i < lc->n; i++) if (lc->slots[i].eid == eid) {
|
||||
m->hits++; lc->slots[i].used = ++m->clock; *out = &lc->slots[i]; return;
|
||||
m->hits++; lc->slots[i].used = ++m->clock; *out = &lc->slots[i];
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
m->miss++;
|
||||
Cfg *c = &m->c;
|
||||
int64_t ng = (int64_t)c->inter * c->hidden, nd = (int64_t)c->hidden * c->inter;
|
||||
Slot *s;
|
||||
if (lc->n < lc->cap) {
|
||||
s = &lc->slots[lc->n++];
|
||||
s->g = malloc(ng); s->u = malloc(ng); s->d = malloc(nd);
|
||||
s->gs = falloc(c->inter); s->us = falloc(c->inter); s->ds = falloc(c->hidden);
|
||||
} else { int lru = 0; for (int i = 1; i < lc->n; i++) if (lc->slots[i].used < lc->slots[lru].used) lru = i; s = &lc->slots[lru]; }
|
||||
float *tmp = falloc(ng > nd ? ng : nd);
|
||||
char nm[256];
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.gate_proj.weight",layer,eid); load_expert_w(m,nm,s->g,s->gs,c->inter,c->hidden,tmp);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.up_proj.weight", layer,eid); load_expert_w(m,nm,s->u,s->us,c->inter,c->hidden,tmp);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.down_proj.weight",layer,eid); load_expert_w(m,nm,s->d,s->ds,c->hidden,c->inter,tmp);
|
||||
free(tmp);
|
||||
s->eid = eid; s->used = ++m->clock;
|
||||
slot_ensure_allocated(m, s);
|
||||
} else {
|
||||
/* LRU eviction — skip pinned and in-flight (eid==-1) slots */
|
||||
int lru = -1;
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].pinned || lc->slots[i].eid < 0) continue;
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
if (lru < 0) {
|
||||
/* All slots are pinned or in-flight; find oldest non-in-flight slot
|
||||
* (may be pinned, but never select one currently being loaded). */
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid < 0) continue; /* never evict in-flight */
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
}
|
||||
if (lru < 0) lru = 0; /* absolute last resort: all in-flight, evict slot 0 */
|
||||
s = &lc->slots[lru];
|
||||
s->pinned = 0;
|
||||
}
|
||||
s->eid = -1;
|
||||
s->used = ++m->clock;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
load_expert_merged(m, layer, eid, s);
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
s->eid = eid;
|
||||
s->pinned = m->is_pinned[layer * c->n_experts + eid];
|
||||
s->used = ++m->clock;
|
||||
*out = s;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
|
||||
/* ---------- IMPROVEMENT 2: pin top-N hot experts per layer ---------- */
|
||||
static void pin_hot_experts(Model *m) {
|
||||
Cfg *c = &m->c;
|
||||
if (m->hot_n <= 0 || m->hot_pinned) return;
|
||||
m->hot_pinned = 1;
|
||||
|
||||
int is_dynamic = (m->hot_n >= 100);
|
||||
double thresh = is_dynamic ? (double)m->hot_n / 1000.0 : 0.0;
|
||||
|
||||
int pinned_total = 0;
|
||||
for (int l = 0; l < c->n_layers; l++) {
|
||||
uint32_t *freq_l = m->freq + (int64_t)l * c->n_experts;
|
||||
|
||||
uint64_t layer_total = 0;
|
||||
for (int e = 0; e < c->n_experts; e++) layer_total += freq_l[e];
|
||||
if (layer_total == 0) continue;
|
||||
|
||||
int max_pin = m->cache[l].cap - 8;
|
||||
if (max_pin < 4) max_pin = 4;
|
||||
|
||||
int hn = is_dynamic ? max_pin : (m->hot_n < c->n_experts ? m->hot_n : c->n_experts);
|
||||
if (hn > 256) hn = 256;
|
||||
int hot_eids[256];
|
||||
int actual_hn = 0;
|
||||
|
||||
for (int k = 0; k < hn; k++) {
|
||||
int best = -1; uint32_t bv = 0;
|
||||
for (int e = 0; e < c->n_experts; e++) {
|
||||
int already = 0;
|
||||
for (int j = 0; j < k; j++) if (hot_eids[j] == e) { already = 1; break; }
|
||||
if (!already && freq_l[e] > bv) { bv = freq_l[e]; best = e; }
|
||||
}
|
||||
if (best < 0 || bv == 0) break;
|
||||
if (is_dynamic && bv < thresh * layer_total) break;
|
||||
hot_eids[k] = best;
|
||||
actual_hn++;
|
||||
}
|
||||
|
||||
for (int k = 0; k < actual_hn; k++) {
|
||||
int eid = hot_eids[k];
|
||||
m->is_pinned[l * c->n_experts + eid] = 1;
|
||||
|
||||
LCache *lc = &m->cache[l];
|
||||
int found = 0;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid == eid) { lc->slots[i].pinned = 1; found = 1; break; }
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
if (!found && g_pilot > 0) {
|
||||
/* Only enqueue when the prefetch worker is active (PILOT>0). */
|
||||
ensure_pilot_worker_started(m);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
int gidx = l * c->n_experts + eid;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
int already = m->is_queued[gidx];
|
||||
if (!already && w - r < 4096) {
|
||||
pilot_q[w & 4095].l = l; pilot_q[w & 4095].e = eid;
|
||||
m->is_queued[gidx] = 1;
|
||||
__atomic_store_n(&pilot_w, w + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
pinned_total++;
|
||||
}
|
||||
}
|
||||
if (is_dynamic) {
|
||||
printf("[HOT] Dynamic Pinned %d experts total (thresh=%.1f%%) after %d warmup tokens\n",
|
||||
pinned_total, thresh * 100.0, m->freq_token_count);
|
||||
} else {
|
||||
printf("[HOT] Pinned %d experts (top-%d/layer) after %d warmup tokens\n",
|
||||
pinned_total, m->hot_n, m->freq_token_count);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* ---------- RoPE su un vettore di una testa (head_dim) a posizione assoluta pos ---------- */
|
||||
static void rope_head(float *x, int pos, const Cfg *c) {
|
||||
int h = c->head_dim / 2;
|
||||
@@ -325,6 +557,19 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
float *g = falloc(I), *u = falloc(I), *hh = falloc(D);
|
||||
for (int s = 0; s < S; s++) {
|
||||
float *pr = logits + (int64_t)s*E;
|
||||
if (m->momentum_logits && m->pilot_smooth > 0.f) {
|
||||
float *ema = m->momentum_logits + (int64_t)layer * E;
|
||||
int is_zero = 1;
|
||||
for (int e = 0; e < E; e++) { if (ema[e] != 0.f) { is_zero = 0; break; } }
|
||||
if (is_zero) {
|
||||
for (int e = 0; e < E; e++) ema[e] = pr[e];
|
||||
} else {
|
||||
for (int e = 0; e < E; e++) {
|
||||
ema[e] = (1.f - m->pilot_smooth) * pr[e] + m->pilot_smooth * ema[e];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
softmax_row(pr, E);
|
||||
/* top-K indici (selezione parziale) */
|
||||
int idx[64]; float val[64];
|
||||
@@ -337,6 +582,11 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
idx[kk] = best; val[kk] = bv;
|
||||
}
|
||||
if (c->norm_topk) { float sm=0; for(int kk=0;kk<K;kk++) sm+=val[kk]; for(int kk=0;kk<K;kk++) val[kk]/=sm; }
|
||||
/* IMPROVEMENT 2: update activation heatmap (before pinning activates) */
|
||||
if (!m->hot_pinned && m->freq) {
|
||||
uint32_t *freq_l = m->freq + (int64_t)layer * E;
|
||||
for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) freq_l[idx[kk]]++;
|
||||
}
|
||||
const float *xs = x + (int64_t)s*D;
|
||||
for (int kk = 0; kk < K; kk++) {
|
||||
Slot *e; expert_get(m, layer, idx[kk], &e);
|
||||
@@ -352,9 +602,20 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
free(logits); free(g); free(u); free(hh);
|
||||
}
|
||||
|
||||
/* un passo: token nuovi ids[S] a posizione pos_base. Ritorna logits dell'ultimo token (malloc'd). */
|
||||
static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
Cfg *c = &m->c; int D = c->hidden;
|
||||
if (g_pilot && m->token_count > 0) {
|
||||
/* Flush stale prefetch requests: clear is_queued so pilot_realload
|
||||
* will skip any entries still sitting in pilot_q for the previous
|
||||
* token. We deliberately do NOT move pilot_w backwards; that would
|
||||
* break the ring-buffer invariant (pilot_r could exceed pilot_w if
|
||||
* the worker consumed an entry concurrently). The worker will drain
|
||||
* the stale slots harmlessly because pilot_realload already exits
|
||||
* early when the expert is already cached or is_queued is clear. */
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
memset(m->is_queued, 0, (size_t)c->n_layers * c->n_experts);
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
float *x = falloc((int64_t)S*D);
|
||||
for (int s = 0; s < S; s++) memcpy(x + (int64_t)s*D, m->embed + (int64_t)ids[s]*D, D*sizeof(float));
|
||||
float *nrm = falloc((int64_t)S*D), *tmp = falloc((int64_t)S*D);
|
||||
@@ -363,12 +624,26 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
for (int s = 0; s < S; s++) rmsnorm_row(nrm + (int64_t)s*D, x + (int64_t)s*D, l->in_ln, D, c->eps);
|
||||
attention(m, l, i, nrm, S, pos_base, tmp);
|
||||
for (int64_t j = 0; j < (int64_t)S*D; j++) x[j] += tmp[j];
|
||||
/* IMPROVEMENT 1: PILOT=1 -> 1-layer lookahead */
|
||||
if (g_pilot >= 1 && S <= 8 && i + 1 < c->n_layers)
|
||||
pilot_prefetch(m, i + 1, x, S);
|
||||
for (int s = 0; s < S; s++) rmsnorm_row(nrm + (int64_t)s*D, x + (int64_t)s*D, l->post_ln, D, c->eps);
|
||||
moe(m, l, i, nrm, S, tmp);
|
||||
for (int64_t j = 0; j < (int64_t)S*D; j++) x[j] += tmp[j];
|
||||
|
||||
/* PREDICTION IMPROVEMENT C (Residual gate trick):
|
||||
* PILOT=2 -> prefetch layer i+2 using completed state x (containing MoE residual). */
|
||||
if (g_pilot >= 2 && S <= 8 && i + 2 < c->n_layers)
|
||||
pilot_prefetch(m, i + 2, x, S);
|
||||
if (g_pilot >= 3 && S <= 8 && i + 3 < c->n_layers)
|
||||
pilot_prefetch(m, i + 3, x, S);
|
||||
|
||||
}
|
||||
/* count actual tokens processed (S>1 during prefill) */
|
||||
m->token_count += S; m->freq_token_count += S;
|
||||
if (!m->hot_pinned && m->hot_n > 0 && m->freq_token_count >= m->warmup_tokens)
|
||||
pin_hot_experts(m);
|
||||
m->kv_len = pos_base + S;
|
||||
/* solo l'ultimo token -> logits */
|
||||
float *last = falloc(D);
|
||||
rmsnorm_row(last, x + (int64_t)(S-1)*D, m->final_norm, D, c->eps);
|
||||
float *logit = falloc(c->vocab);
|
||||
@@ -377,6 +652,192 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
return logit;
|
||||
}
|
||||
|
||||
static void pilot_realload(Model *m, int layer, int eid) {
|
||||
LCache *lc = &m->cache[layer];
|
||||
Cfg *c = &m->c;
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
/* Early-exit if entry was flushed (is_queued cleared) while waiting. */
|
||||
if (!m->is_queued[layer * c->n_experts + eid]) {
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid == eid) {
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
}
|
||||
Slot *s;
|
||||
if (lc->n < lc->cap) {
|
||||
s = &lc->slots[lc->n++];
|
||||
slot_ensure_allocated(m, s);
|
||||
} else {
|
||||
/* LRU eviction — skip pinned and in-flight (eid==-1) slots */
|
||||
int lru = -1;
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].pinned || lc->slots[i].eid < 0) continue;
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
if (lru < 0) {
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return; /* all pinned/in-flight, skip */
|
||||
}
|
||||
s = &lc->slots[lru]; s->pinned = 0;
|
||||
}
|
||||
s->eid = -1; s->used = ++m->clock;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
load_expert_merged(m, layer, eid, s);
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
s->eid = eid;
|
||||
s->pinned = m->is_pinned[layer * c->n_experts + eid];
|
||||
s->used = ++m->clock;
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
|
||||
static void *pilot_worker(void *arg) {
|
||||
(void)arg;
|
||||
while (1) {
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_ACQUIRE);
|
||||
if (r == w) {
|
||||
sleep_ms(1);
|
||||
continue;
|
||||
}
|
||||
int layer = pilot_q[r & 4095].l;
|
||||
int eid = pilot_q[r & 4095].e;
|
||||
pilot_realload(pilot_m, layer, eid);
|
||||
__atomic_store_n(&pilot_r, r + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void pilot_prefetch(Model *m, int lnext, const float *x, int S) {
|
||||
if (lnext < 0 || lnext >= m->c.n_layers) return;
|
||||
Cfg *c = &m->c; int D = c->hidden, E = c->n_experts;
|
||||
ensure_pilot_worker_started(m);
|
||||
float *logits = falloc((int64_t)S * E);
|
||||
Layer *l = &m->L[lnext];
|
||||
|
||||
// PREDICTION IMPROVEMENT B: Apply RMSNorm to x using destination layer's post_ln
|
||||
// This scales inputs to the distribution expected by l->gate.
|
||||
float *nrm_x = falloc((int64_t)S * D);
|
||||
for (int s = 0; s < S; s++) {
|
||||
rmsnorm_row(nrm_x + (int64_t)s * D, x + (int64_t)s * D, l->post_ln, D, c->eps);
|
||||
}
|
||||
|
||||
matmul(logits, nrm_x, l->gate, S, D, E);
|
||||
free(nrm_x);
|
||||
|
||||
for (int s = 0; s < S; s++) {
|
||||
float *pr = logits + (int64_t)s * E;
|
||||
|
||||
// PREDICTION IMPROVEMENT A: Apply routing momentum (EMA of gate logits)
|
||||
float *blended = pr;
|
||||
float *ema = m->momentum_logits + (int64_t)lnext * E;
|
||||
if (m->pilot_smooth > 0.f) {
|
||||
blended = falloc(E);
|
||||
int is_zero = 1;
|
||||
for (int e = 0; e < E; e++) { if (ema[e] != 0.f) { is_zero = 0; break; } }
|
||||
if (is_zero) {
|
||||
for (int e = 0; e < E; e++) {
|
||||
ema[e] = pr[e];
|
||||
blended[e] = pr[e];
|
||||
}
|
||||
} else {
|
||||
for (int e = 0; e < E; e++) {
|
||||
blended[e] = (1.f - m->pilot_smooth) * pr[e] + m->pilot_smooth * ema[e];
|
||||
ema[e] = blended[e]; // update EMA
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int cand = 0;
|
||||
int idx[128];
|
||||
|
||||
float max_logit = -1e30f;
|
||||
for (int e = 0; e < E; e++) { if (blended[e] > max_logit) max_logit = blended[e]; }
|
||||
float *exps = falloc(E);
|
||||
float sum_exps = 0.f;
|
||||
for (int e = 0; e < E; e++) {
|
||||
exps[e] = expf(blended[e] - max_logit);
|
||||
sum_exps += exps[e];
|
||||
}
|
||||
|
||||
float cum_sum = 0.f;
|
||||
int min_cand = c->topk;
|
||||
int max_cand = c->topk * g_wide;
|
||||
if (max_cand < min_cand) max_cand = min_cand;
|
||||
if (max_cand > 128) max_cand = 128; /* idx[] buffer bound */
|
||||
if (max_cand > E) max_cand = E;
|
||||
|
||||
for (int kk = 0; kk < max_cand; kk++) {
|
||||
int best = -1; float bv = -1.f;
|
||||
for (int e = 0; e < E; e++) {
|
||||
int taken = 0; for (int j = 0; j < kk; j++) if (idx[j] == e) { taken=1; break; }
|
||||
if (!taken && exps[e] > bv) { bv = exps[e]; best = e; }
|
||||
}
|
||||
if (best < 0) break;
|
||||
idx[kk] = best;
|
||||
cum_sum += bv;
|
||||
cand++;
|
||||
if (cum_sum >= m->pilot_conf_limit * sum_exps && cand >= min_cand) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(exps);
|
||||
|
||||
if (blended != pr) free(blended);
|
||||
|
||||
/* IMPROVEMENT 5: sort candidates by eid for sequential SSD read locality */
|
||||
for (int a = 0; a < cand-1; a++)
|
||||
for (int b = a+1; b < cand; b++)
|
||||
if (idx[b] >= 0 && (idx[a] < 0 || idx[a] > idx[b])) { int t = idx[a]; idx[a] = idx[b]; idx[b] = t; }
|
||||
|
||||
for (int kk = 0; kk < cand; kk++) {
|
||||
int eid = idx[kk];
|
||||
if (eid < 0) continue;
|
||||
int found = 0;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
LCache *lc = &m->cache[lnext];
|
||||
for (int z = 0; z < lc->n; z++) {
|
||||
if (lc->slots[z].eid == eid) { found = 1; break; }
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
if (!found) {
|
||||
int gidx = lnext * E + eid;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
int already_queued = m->is_queued[gidx];
|
||||
if (!already_queued) {
|
||||
m->is_queued[gidx] = 1;
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
if (!already_queued) {
|
||||
unsigned w2 = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r2 = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
if (w2 - r2 < 4096) {
|
||||
pilot_q[w2 & 4095].l = lnext;
|
||||
pilot_q[w2 & 4095].e = eid;
|
||||
__atomic_store_n(&pilot_w, w2 + 1, __ATOMIC_RELEASE);
|
||||
} else {
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
m->is_queued[gidx] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
free(logits);
|
||||
}
|
||||
|
||||
|
||||
/* generazione greedy. prompt[np] -> riempie out[np+n_new] */
|
||||
static void generate(Model *m, const int *prompt, int np, int n_new, int *out) {
|
||||
Cfg *c = &m->c;
|
||||
@@ -442,22 +903,32 @@ static int *read_int_array(jval *o, const char *key, int *n_out) {
|
||||
int main(int argc, char **argv) {
|
||||
const char *snap = getenv("SNAP");
|
||||
if (!snap) { fprintf(stderr, "set SNAP=<snapshot directory>\n"); return 1; }
|
||||
int cap = argc > 1 ? atoi(argv[1]) : 16;
|
||||
int bits = argc > 2 ? atoi(argv[2]) : 8;
|
||||
if (bits < 2 || bits > 8) { /* expert storage is int8_t: bits>8 truncates in quantize_rows (#134). f32 mode is not implemented here — int8 is already token-exact vs the oracle. */
|
||||
fprintf(stderr, "quant_bits must be 2..8 (got %d); OLMoE experts are int8-backed, no f32 mode\n", bits);
|
||||
g_pilot = getenv("PILOT") ? atoi(getenv("PILOT")) : 0;
|
||||
g_wide = getenv("WIDE") ? atoi(getenv("WIDE")) : 1;
|
||||
if (g_wide < 1) g_wide = 1;
|
||||
if (g_wide > 4) g_wide = 4;
|
||||
int hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0;
|
||||
int cap = argc > 1 ? atoi(argv[1]) : 16;
|
||||
int bits = argc > 2 ? atoi(argv[2]) : 8;
|
||||
if (bits < 2 || bits > 8) {
|
||||
fprintf(stderr, "quant_bits must be 2..8 (got %d)\n", bits);
|
||||
return 1;
|
||||
}
|
||||
const char *refpath = argc > 3 ? argv[3] : "ref.json";
|
||||
|
||||
FILE *f = fopen(refpath, "rb"); if(!f){perror(refpath);return 1;}
|
||||
float smooth = getenv("SMOOTH") ? (float)atof(getenv("SMOOTH")) : 0.3f;
|
||||
float conf = getenv("CONF_LIMIT") ? (float)atof(getenv("CONF_LIMIT")) : 0.92f;
|
||||
|
||||
printf("== Streaming C engine v2.2 | cache=%d/layer bits=%d pilot=%d wide=%d hot=%d smooth=%.2f conf=%.2f ==\n",
|
||||
cap, bits, g_pilot, g_wide, hot_n, smooth, conf);
|
||||
|
||||
FILE *f = fopen(refpath, "rb"); if (!f) { perror(refpath); return 1; }
|
||||
fseek(f,0,SEEK_END); long n=ftell(f); fseek(f,0,SEEK_SET);
|
||||
char *buf=malloc(n+1); if(fread(buf,1,n,f)!=(size_t)n){} buf[n]=0; fclose(f);
|
||||
char *buf=malloc(n+1); if (fread(buf,1,n,f)!=(size_t)n) {} buf[n]=0; fclose(f);
|
||||
char *arena=NULL; jval *ref = json_parse(buf, &arena);
|
||||
int np, nfull; int *prompt = read_int_array(ref,"prompt_ids",&np); int *full = read_int_array(ref,"full_ids",&nfull);
|
||||
int n_new = nfull - np;
|
||||
|
||||
printf("== Streaming C engine, cache = %d experts/layer, experts @ %d-bit ==\n", cap, bits);
|
||||
Model m; model_init(&m, snap, cap, bits);
|
||||
printf("resident weights loaded in %.1fs | RSS after load: %.2f GB\n", m.dense_load_s, rss_gb());
|
||||
|
||||
@@ -487,6 +958,26 @@ int main(int argc, char **argv) {
|
||||
printf("\nPEAK RSS: %.2f GB\n", rss_gb());
|
||||
printf("Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0,
|
||||
(unsigned long long)m.hits, (unsigned long long)m.miss);
|
||||
|
||||
|
||||
// Persistent Hot Pinning: save dynamic pinning if newly created
|
||||
if (m.hot_pinned) {
|
||||
char pinpath[512];
|
||||
snprintf(pinpath, sizeof(pinpath), "%s/hot_pinned.bin", snap);
|
||||
FILE *pinf_chk = fopen(pinpath, "rb");
|
||||
if (!pinf_chk) {
|
||||
FILE *pinf_save = fopen(pinpath, "wb");
|
||||
if (pinf_save) {
|
||||
size_t expected_size = (size_t)m.c.n_layers * m.c.n_experts;
|
||||
fwrite(m.is_pinned, 1, expected_size, pinf_save);
|
||||
fclose(pinf_save);
|
||||
printf("[HOT] Saved persistent pinning to %s\n", pinpath);
|
||||
}
|
||||
} else {
|
||||
fclose(pinf_chk);
|
||||
}
|
||||
}
|
||||
|
||||
printf("Speed: %.2f tok/s (%.1fs for %d tokens)\n", n_new/dt, dt, n_new);
|
||||
free(buf); free(arena);
|
||||
return 0;
|
||||
|
||||
+85
-16
@@ -370,10 +370,46 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No
|
||||
return "".join(prompt)
|
||||
|
||||
|
||||
# Generic whitespace-tolerant JSON grammar for response_format {"type": "json_object"}.
|
||||
# Draft-source semantics: positions with one legal byte draft; jws points just keep
|
||||
# the walker alive through the model's own spacing (see docs/grammar-draft.md).
|
||||
GENERIC_JSON_GBNF = (
|
||||
'root ::= jws jval jws\n'
|
||||
'jval ::= jobj | jarr | jstr | jnum | "true" | "false" | "null"\n'
|
||||
'jobj ::= "{" jws ( jstr jws ":" jws jval jws ( "," jws jstr jws ":" jws jval jws )* )? "}"\n'
|
||||
'jarr ::= "[" jws ( jval jws ( "," jws jval jws )* )? "]"\n'
|
||||
'jstr ::= "\\"" jchar* "\\""\n'
|
||||
'jchar ::= [^"\\\\\\x00-\\x1f] | "\\\\" ( ["\\\\/bfnrt] | "u" jhex jhex jhex jhex )\n'
|
||||
'jhex ::= [0-9a-fA-F]\n'
|
||||
'jnum ::= "-"? ( "0" | [1-9] [0-9]* ) ( "." [0-9]+ )? ( ( "e" | "E" ) ( "+" | "-" )? [0-9]+ )?\n'
|
||||
'jws ::= ( " " | "\\t" | "\\n" | "\\r" )*\n'
|
||||
)
|
||||
|
||||
def generation_options(body, limit):
|
||||
if body.get("n", 1) != 1:
|
||||
raise APIError(400, "Colibri currently supports `n=1` only.", "n", "unsupported_value")
|
||||
# `tools`/`functions` are handled by render_chat (declaration) + parse_tool_calls (output).
|
||||
# Validate tools/functions structure early so malformed input fails with a clear error.
|
||||
tools_raw = body.get("tools") or body.get("functions")
|
||||
if tools_raw is not None:
|
||||
if not isinstance(tools_raw, list):
|
||||
raise APIError(400, "`tools` must be a non-empty array.", "tools", "invalid_value")
|
||||
if not tools_raw:
|
||||
raise APIError(400, "`tools` must be a non-empty array.", "tools", "invalid_value")
|
||||
for idx, tool in enumerate(tools_raw):
|
||||
if not isinstance(tool, dict):
|
||||
raise APIError(400, f"Each tool must be an object, got {type(tool).__name__} at index {idx}.",
|
||||
f"tools.{idx}", "invalid_value")
|
||||
fn = tool.get("function", tool) if isinstance(tool, dict) else {}
|
||||
if not isinstance(fn, dict):
|
||||
raise APIError(400, f"Tool function must be an object at index {idx}.",
|
||||
f"tools.{idx}.function", "invalid_value")
|
||||
if not fn.get("name"):
|
||||
raise APIError(400, f"Each tool must have a `name` at index {idx}.",
|
||||
f"tools.{idx}.function.name", "invalid_value")
|
||||
if not isinstance(fn["name"], str):
|
||||
raise APIError(400, f"Tool `name` must be a string at index {idx}.",
|
||||
f"tools.{idx}.function.name", "invalid_value")
|
||||
choice = body.get("tool_choice")
|
||||
if choice is not None:
|
||||
if isinstance(choice, str):
|
||||
@@ -403,10 +439,38 @@ def generation_options(body, limit):
|
||||
raise APIError(400, "Token penalties are not supported yet.", None, "unsupported_parameter")
|
||||
if body.get("seed") is not None:
|
||||
raise APIError(400, "Per-request seeds are not supported yet.", "seed", "unsupported_parameter")
|
||||
# response_format -> optional per-request grammar for the engine's grammar-forced
|
||||
# draft source (#70/#148). NEVER a sampling constraint: drafts are verified, so a
|
||||
# schema the engine cannot compile degrades to "no speedup", not to an error and
|
||||
# not to changed output. json_schema payloads are forwarded as-is (the engine
|
||||
# compiles them via schema_gbnf.h); {"type": "gbnf"} is a raw-GBNF extension.
|
||||
grammar = None
|
||||
response_format = body.get("response_format")
|
||||
if response_format not in (None, {"type": "text"}):
|
||||
raise APIError(400, "Only the default text response format is supported.",
|
||||
"response_format", "unsupported_parameter")
|
||||
if response_format is not None and response_format != {"type": "text"}:
|
||||
if not isinstance(response_format, dict) or "type" not in response_format:
|
||||
raise APIError(400, "`response_format` must be an object with a `type`.",
|
||||
"response_format", "invalid_value")
|
||||
ftype = response_format["type"]
|
||||
if ftype == "json_object":
|
||||
grammar = GENERIC_JSON_GBNF
|
||||
elif ftype == "json_schema":
|
||||
schema = (response_format.get("json_schema") or {}).get("schema")
|
||||
if not isinstance(schema, dict):
|
||||
raise APIError(400, "`response_format.json_schema.schema` must be an object.",
|
||||
"response_format", "invalid_value")
|
||||
grammar = json.dumps(schema)
|
||||
elif ftype == "gbnf":
|
||||
grammar = response_format.get("grammar")
|
||||
if not isinstance(grammar, str) or not grammar.strip():
|
||||
raise APIError(400, "`response_format.grammar` must be a non-empty GBNF string.",
|
||||
"response_format", "invalid_value")
|
||||
else:
|
||||
raise APIError(400, "`response_format.type` must be \"text\", \"json_object\", "
|
||||
"\"json_schema\" or \"gbnf\".",
|
||||
"response_format", "unsupported_value")
|
||||
if grammar is not None and len(grammar.encode("utf-8")) > (1 << 20):
|
||||
raise APIError(400, "`response_format` grammar/schema exceeds 1 MiB.",
|
||||
"response_format", "invalid_value")
|
||||
|
||||
maximum = body.get("max_completion_tokens")
|
||||
maximum_param = "max_completion_tokens"
|
||||
@@ -433,7 +497,7 @@ def generation_options(body, limit):
|
||||
if (isinstance(top_p, bool) or not isinstance(top_p, (int, float)) or
|
||||
not math.isfinite(top_p) or not 0 < top_p <= 1):
|
||||
raise APIError(400, "`top_p` must be greater than 0 and at most 1.", "top_p")
|
||||
return maximum, float(temperature), float(top_p)
|
||||
return maximum, float(temperature), float(top_p), grammar
|
||||
|
||||
|
||||
def read_engine_turn(stream, sentinel, on_bytes):
|
||||
@@ -597,12 +661,15 @@ class Engine:
|
||||
self._fail_pending(error)
|
||||
|
||||
def generate(self, prompt, max_tokens, temperature, top_p, on_text, cache_slot=0,
|
||||
cancelled=None):
|
||||
cancelled=None, grammar=None):
|
||||
if isinstance(cache_slot, bool) or not isinstance(cache_slot, int) or not 0 <= cache_slot < self.kv_slots:
|
||||
raise APIError(400, "Invalid cache slot.", "cache_slot")
|
||||
payload = prompt.encode("utf-8")
|
||||
if b"\0" in payload:
|
||||
raise APIError(400, "NUL bytes are not supported in prompts.", "messages")
|
||||
gpayload = grammar.encode("utf-8") if grammar else b""
|
||||
if b"\0" in gpayload:
|
||||
raise APIError(400, "NUL bytes are not supported in grammars.", "response_format")
|
||||
decoder = codecs.getincrementaldecoder("utf-8")("replace")
|
||||
|
||||
def decode(data):
|
||||
@@ -622,12 +689,13 @@ class Engine:
|
||||
self.next_request_id += 1
|
||||
self.pending[request_id] = events
|
||||
header = (f"SUBMIT {request_id} {cache_slot} {len(payload)} {max_tokens} "
|
||||
f"{temperature:.8g} {top_p:.8g}\n").encode()
|
||||
f"{temperature:.8g} {top_p:.8g}"
|
||||
+ (f" {len(gpayload)}" if gpayload else "") + "\n").encode()
|
||||
try:
|
||||
with self.write_lock:
|
||||
if self.process.poll() is not None:
|
||||
raise RuntimeError("colibri engine is not running")
|
||||
self.process.stdin.write(header + payload + b"\n")
|
||||
self.process.stdin.write(header + payload + gpayload + b"\n")
|
||||
self.process.stdin.flush()
|
||||
except Exception:
|
||||
with self.pending_lock:
|
||||
@@ -863,7 +931,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def generation(self, body, prompt, request_id, chat):
|
||||
def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=None):
|
||||
# COLI_DEBUG tees the engine transaction to stderr: 1 = decoded output stream only,
|
||||
# 2 = both sides (rendered prompt + output). render_chat already folds prior turns and
|
||||
# tool results into `prompt`, so level 2 is the full conversation the engine saw.
|
||||
@@ -874,9 +942,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
if dbg >= 2:
|
||||
sys.stderr.write(f"\n===== PROMPT [{request_id}] =====\n{prompt}\n===== OUTPUT [{request_id}] =====\n")
|
||||
sys.stderr.flush()
|
||||
maximum, temperature, top_p = generation_options(body, self.server.max_tokens)
|
||||
tools = (body.get("tools") or body.get("functions") or None) if chat else None
|
||||
if body.get("tool_choice") == "none":
|
||||
maximum, temperature, top_p, grammar = generation_options(body, self.server.max_tokens)
|
||||
# tools and tool_choice come from chat_completion() already processed/filtered
|
||||
if chat and tool_choice == "none":
|
||||
tools = None # client forbade tools: never surface tool_calls
|
||||
cache_slot = body.get("cache_slot")
|
||||
if (cache_slot is not None and
|
||||
@@ -903,7 +971,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
output = []
|
||||
stats = self.server.engine.generate(
|
||||
prompt, maximum, temperature, top_p, output.append, cache_slot,
|
||||
self.client_disconnected)
|
||||
self.client_disconnected, grammar=grammar)
|
||||
text = "".join(output)
|
||||
length_finish = "length" if stats["length_limited"] else "stop"
|
||||
if chat and tools:
|
||||
@@ -1010,7 +1078,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
sp["buf"] = sp["buf"][flush:]
|
||||
stats = self.server.engine.generate(
|
||||
prompt, maximum, temperature, top_p, emit_tools, cache_slot,
|
||||
lambda: not connected)
|
||||
lambda: not connected, grammar=grammar)
|
||||
if not sp["tool"] and sp["buf"]:
|
||||
emit(sp["buf"]) # no tool call happened: flush held tail
|
||||
_content, calls = parse_tool_calls("".join(raw), tools)
|
||||
@@ -1027,7 +1095,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
emit(chunk)
|
||||
stats = self.server.engine.generate(
|
||||
prompt, maximum, temperature, top_p, emit_plain, cache_slot,
|
||||
lambda: not connected)
|
||||
lambda: not connected, grammar=grammar)
|
||||
finish = "length" if stats["length_limited"] else "stop"
|
||||
ka_stop.set() # generation done: stop the keepalive pump
|
||||
ka_thread.join(timeout=2)
|
||||
@@ -1078,9 +1146,10 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
if not isinstance(enable_thinking, bool):
|
||||
raise APIError(400, "`enable_thinking` must be a boolean.", "enable_thinking")
|
||||
tools = body.get("tools") or body.get("functions") or None
|
||||
tool_choice = body.get("tool_choice")
|
||||
prompt = render_chat(body.get("messages"), enable_thinking, reasoning_effort, tools,
|
||||
body.get("tool_choice"))
|
||||
self.generation(body, prompt, request_id, True)
|
||||
tool_choice)
|
||||
self.generation(body, prompt, request_id, True, tools, tool_choice)
|
||||
|
||||
def completion(self, body, request_id):
|
||||
prompt = body.get("prompt")
|
||||
|
||||
@@ -0,0 +1,796 @@
|
||||
/* quant.h — quantized matmul kernels (header-only, all functions static).
|
||||
* Multi-architecture SIMD: AVX2 / AVX-512 / AVX-VNNI / ARM NEON / NEON-SDOT /
|
||||
* NEON-i8mm / POWER VSX. Pure compute — no Model or QT dependency. */
|
||||
#ifndef COLI_QUANT_H
|
||||
#define COLI_QUANT_H
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <math.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef _OPENMP
|
||||
#include <omp.h>
|
||||
#endif
|
||||
|
||||
/* ---- SIMD includes -------------------------------------------------------- */
|
||||
#ifdef __AVX2__
|
||||
#include <immintrin.h>
|
||||
static inline float hsum256(__m256 v){
|
||||
__m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1);
|
||||
lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh);
|
||||
sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo);
|
||||
}
|
||||
static inline int hsum256_i32(__m256i v){
|
||||
__m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1);
|
||||
lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo);
|
||||
return _mm_cvtsi128_si32(lo);
|
||||
}
|
||||
#endif
|
||||
#if defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
static inline int hsum128_i32(__m128i v){
|
||||
v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v);
|
||||
}
|
||||
#endif
|
||||
#ifdef __ARM_NEON
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
#ifdef __VSX__
|
||||
#include <altivec.h>
|
||||
#undef vector
|
||||
#undef pixel
|
||||
#undef bool
|
||||
#endif
|
||||
|
||||
/* ---- AVX-512 int4->float accumulator -------------------------------------- */
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
static int g_i4_acc512=1;
|
||||
static inline float dot_i4f_avx512(const uint8_t *w,const float *x,int I){
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m512i b8=_mm512_set1_epi32(8);
|
||||
__m512 acc0=_mm512_setzero_ps(),acc1=_mm512_setzero_ps(); int i=0;
|
||||
for(;i+32<=I;i+=32){ __m128i by=_mm_loadu_si128((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi),n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m512 w0=_mm512_cvtepi32_ps(_mm512_sub_epi32(_mm512_cvtepu8_epi32(n0),b8));
|
||||
__m512 w1=_mm512_cvtepi32_ps(_mm512_sub_epi32(_mm512_cvtepu8_epi32(n1),b8));
|
||||
acc0=_mm512_fmadd_ps(_mm512_loadu_ps(x+i),w0,acc0);
|
||||
acc1=_mm512_fmadd_ps(_mm512_loadu_ps(x+i+16),w1,acc1);
|
||||
}
|
||||
return _mm512_reduce_add_ps(_mm512_add_ps(acc0,acc1));
|
||||
}
|
||||
static int i4_acc512_selftest(void){
|
||||
enum { N=224 }; uint8_t w[(N+1)/2]; float x[N];
|
||||
for(int i=0;i<N;i++){
|
||||
int q=((i*13+5)&15)-8;
|
||||
if(!(i&1)) w[i>>1]=(uint8_t)(q+8);
|
||||
else w[i>>1]|=(uint8_t)((q+8)<<4);
|
||||
x[i]=(float)(((i*29+7)%101)-50)/37.f;
|
||||
}
|
||||
for(int n=32;n<=N;n+=32){
|
||||
float ref=0; for(int i=0;i<n;i++) ref+=x[i]*(float)(((w[i>>1]>>((i&1)*4))&15)-8);
|
||||
float got=dot_i4f_avx512(w,x,n),tol=2e-5f*(1.f+fabsf(ref));
|
||||
if(fabsf(got-ref)>tol){ fprintf(stderr,"AVX512 i4 selftest n=%d: %.9g != %.9g\n",n,got,ref); return 0; }
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W[O,I] f32 ---------------------------------- */
|
||||
static void matmul(float *y, const float *x, const float *W, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const float *w=W+(int64_t)o*I;
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; for(int i=0;i<I;i++) a+=xs[i]*w[i]; y[(int64_t)s*O+o]=a; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int8 per-row + scale[O] ------------------- */
|
||||
static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#ifdef __AVX2__
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+8<=I;i+=8){ __m256i wi=_mm256_cvtepi8_epi32(_mm_loadl_epi64((const __m128i*)(w+i)));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), _mm256_cvtepi32_ps(wi), acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+8<=I;i+=8){ int16x8_t w16=vmovl_s8(vld1_s8(w+i));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w16))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w16)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
for(;i<I;i++) a+=xs[i]*(float)w[i]; y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int4 packed (2/byte) + scale[O] ------------ */
|
||||
static void matmul_i4(float *y, const float *x, const uint8_t *q4, const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
if(g_i4_acc512){ a=dot_i4f_avx512(w,xs,I); i=I&~31; }
|
||||
else {
|
||||
#endif
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m4=vdup_n_u8(0x0F); const int8x8_t b8=vdup_n_s8(8);
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint8x8_t by=vld1_u8(w+(i>>1));
|
||||
uint8x8x2_t z=vzip_u8(vand_u8(by,m4), vshr_n_u8(by,4));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u8(z.val[0]),b8));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u8(z.val[1]),b8));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i+8), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+12), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
}
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t byte=w[i>>1]; int lo=(int)(byte&0xF)-8, hi=(int)(byte>>4)-8;
|
||||
a += xs[i]*(float)lo + xs[i+1]*(float)hi; }
|
||||
if(i<I){ uint8_t byte=w[i>>1]; int lo=(int)(byte&0xF)-8; a += xs[i]*(float)lo; }
|
||||
y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int4 packed + per-GROUP scales (fmt=4) ----- */
|
||||
static void matmul_i4_grouped(float *y, const float *x, const uint8_t *q4, const float *scale,
|
||||
int S, int I, int O, int gs){
|
||||
int rb=(I+1)/2; int ng=(I+gs-1)/gs;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){
|
||||
const uint8_t *w=q4+(int64_t)o*rb;
|
||||
const float *scl=scale+(int64_t)o*ng;
|
||||
for(int s=0;s<S;s++){
|
||||
const float *xs=x+(int64_t)s*I; float a=0;
|
||||
for(int g=0; g*gs<I; g++){
|
||||
int base=g*gs; int glen=gs; if(base+glen>I) glen=I-base;
|
||||
float sc=scl[g];
|
||||
int i=base;
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(; i+16<=base+glen; i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a+=hsum256(acc)*sc;
|
||||
#endif
|
||||
for(; i<base+glen; i+=2){
|
||||
if(i+1<base+glen){ uint8_t byte=w[i>>1];
|
||||
a+=(xs[i]*(float)((int)(byte&0xF)-8)+xs[i+1]*(float)((int)(byte>>4)-8))*sc; }
|
||||
else { uint8_t byte=w[i>>1]; a+=xs[i]*(float)((int)(byte&0xF)-8)*sc; }
|
||||
}
|
||||
}
|
||||
y[(int64_t)s*O+o]=a;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- fused gate+up: one OMP dispatch for both matrices -------------------- */
|
||||
static void matmul_i4_pair(float *yg, float *yu, const float *x,
|
||||
const uint8_t *qg, const float *sg,
|
||||
const uint8_t *qu, const float *su, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int z=0;z<2*O;z++){
|
||||
int o=z<O?z:z-O; const uint8_t *w=(z<O?qg:qu)+(int64_t)o*rb;
|
||||
float a=0; int i=0;
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
if(g_i4_acc512){ a=dot_i4f_avx512(w,x,I); i=I&~31; }
|
||||
else {
|
||||
#endif
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(x+i),w0,acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(x+i+8),w1,acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m4=vdup_n_u8(0x0F); const int8x8_t b8=vdup_n_s8(8);
|
||||
float32x4_t ac0=vdupq_n_f32(0),ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint8x8_t by=vld1_u8(w+(i>>1));
|
||||
uint8x8x2_t n=vzip_u8(vand_u8(by,m4),vshr_n_u8(by,4));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u8(n.val[0]),b8));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u8(n.val[1]),b8));
|
||||
ac0=vfmaq_f32(ac0,vld1q_f32(x+i),vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1,vld1q_f32(x+i+4),vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0,vld1q_f32(x+i+8),vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1,vld1q_f32(x+i+12),vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
}
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t b=w[i>>1]; a+=x[i]*(float)((b&15)-8)+x[i+1]*(float)((b>>4)-8); }
|
||||
if(i<I) a+=x[i]*(float)((w[i>>1]&15)-8);
|
||||
(z<O?yg:yu)[o]=a*(z<O?sg:su)[o];
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int2 packed (4/byte) + scale[O] ------------ */
|
||||
static void matmul_i2(float *y, const float *x, const uint8_t *q2, const float *scale, int S, int I, int O){
|
||||
int rb=(I+3)/4;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const uint8_t *w=q2+(int64_t)o*rb; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#ifdef __AVX2__
|
||||
const __m128i m2=_mm_set1_epi8(0x03); const __m256i b2=_mm256_set1_epi32(2);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_cvtsi32_si128(*(const int*)(w+(i>>2)));
|
||||
__m128i p0=_mm_and_si128(by,m2), p1=_mm_and_si128(_mm_srli_epi16(by,2),m2);
|
||||
__m128i p2=_mm_and_si128(_mm_srli_epi16(by,4),m2), p3=_mm_and_si128(_mm_srli_epi16(by,6),m2);
|
||||
__m128i lo=_mm_unpacklo_epi8(p0,p1), hi=_mm_unpacklo_epi8(p2,p3);
|
||||
__m128i nib=_mm_unpacklo_epi16(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b2));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b2));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m2v=vdup_n_u8(3); const int8x8_t b2v=vdup_n_s8(2);
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint32_t wd; memcpy(&wd, w+(i>>2), 4);
|
||||
uint8x8_t by=vreinterpret_u8_u32(vdup_n_u32(wd));
|
||||
uint8x8x2_t z01=vzip_u8(vand_u8(by,m2v), vand_u8(vshr_n_u8(by,2),m2v));
|
||||
uint8x8x2_t z23=vzip_u8(vand_u8(vshr_n_u8(by,4),m2v), vshr_n_u8(by,6));
|
||||
uint16x4x2_t zz=vzip_u16(vreinterpret_u16_u8(z01.val[0]), vreinterpret_u16_u8(z23.val[0]));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u16(zz.val[0]),b2v));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u16(zz.val[1]),b2v));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i+8), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+12), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
for(;i<I;i++){ uint8_t byte=w[i>>2]; int sh=(i&3)*2; a += xs[i]*(float)((int)((byte>>sh)&3)-2); }
|
||||
y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- int3-g64 (fmt=5): 3-bit weights with ONE f32 scale per 64-input group -
|
||||
* Per group: 16B low plane (2 bits/val, int2 layout) + 8B high plane (1 bit/val),
|
||||
* values in [-4,3] stored v+4. 3.5 bits/weight effective — the quality/size point
|
||||
* the #132 OLMoE ablation measured BEATING per-row int4. */
|
||||
#define I3_GROUP 64
|
||||
#define I3_GBYTES 24 /* 16B low plane + 8B high plane per group */
|
||||
static inline int64_t i3_groups(int I){ return ((int64_t)I + I3_GROUP - 1) / I3_GROUP; }
|
||||
static inline int64_t i3_rowbytes(int I){ return i3_groups(I) * I3_GBYTES; }
|
||||
|
||||
/* Dequant-on-use with PER-GROUP scale. Exact f32 path only (no IDOT in v1: int8
|
||||
* activations don't compose with per-group accumulation without a kernel
|
||||
* restructure — follow-up). NEON: low plane = matmul_i2's unpack, high plane
|
||||
* expanded via vtst on bit masks; x86 stays scalar for now (follow-up). */
|
||||
static void matmul_i3(float *y, const float *x, const uint8_t *q3, const float *scale, int S, int I, int O){
|
||||
int64_t ng=i3_groups(I), rb=i3_rowbytes(I);
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){
|
||||
const uint8_t *wrow=q3+(int64_t)o*rb;
|
||||
const float *srow=scale+(int64_t)o*ng;
|
||||
for(int s=0;s<S;s++){
|
||||
const float *xs=x+(int64_t)s*I;
|
||||
float acc=0;
|
||||
for(int64_t g=0; g<ng; g++){
|
||||
const uint8_t *lo=wrow+g*I3_GBYTES, *hi=lo+16;
|
||||
int base=(int)(g*I3_GROUP), n = I-base < I3_GROUP ? I-base : I3_GROUP;
|
||||
float a=0; int k=0;
|
||||
#if defined(__ARM_NEON)
|
||||
if(n==I3_GROUP){
|
||||
const uint8x8_t m2v=vdup_n_u8(3); const int8x16_t b4q=vdupq_n_s8(4);
|
||||
const uint8x16_t bitm={1,2,4,8,16,32,64,128,1,2,4,8,16,32,64,128};
|
||||
const uint8x16_t fourq=vdupq_n_u8(4);
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;k+16<=I3_GROUP;k+=16){
|
||||
uint32_t wd; memcpy(&wd, lo+(k>>2), 4); /* 4 bytes = 16 low-plane values */
|
||||
uint8x8_t by=vreinterpret_u8_u32(vdup_n_u32(wd));
|
||||
uint8x8x2_t z01=vzip_u8(vand_u8(by,m2v), vand_u8(vshr_n_u8(by,2),m2v));
|
||||
uint8x8x2_t z23=vzip_u8(vand_u8(vshr_n_u8(by,4),m2v), vshr_n_u8(by,6));
|
||||
uint16x4x2_t zz=vzip_u16(vreinterpret_u16_u8(z01.val[0]), vreinterpret_u16_u8(z23.val[0]));
|
||||
uint8x16_t lov=vcombine_u8(vreinterpret_u8_u16(zz.val[0]), vreinterpret_u8_u16(zz.val[1]));
|
||||
uint8x16_t hv=vcombine_u8(vdup_n_u8(hi[k>>3]), vdup_n_u8(hi[(k>>3)+1]));
|
||||
uint8x16_t hb=vandq_u8(vtstq_u8(hv,bitm), fourq); /* 4 where high bit set */
|
||||
int8x16_t wq=vsubq_s8(vreinterpretq_s8_u8(vaddq_u8(lov,hb)), b4q); /* [-4,3] in order */
|
||||
int16x8_t w0=vmovl_s8(vget_low_s8(wq)), w1=vmovl_s8(vget_high_s8(wq));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+base+k), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+base+k+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+base+k+8), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+base+k+12), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1))));
|
||||
}
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
}
|
||||
#endif
|
||||
for(;k<n;k++){
|
||||
unsigned u=((lo[k>>2]>>((k&3)*2))&3) | (((hi[k>>3]>>(k&7))&1)<<2);
|
||||
a += xs[base+k]*(float)((int)u-4);
|
||||
}
|
||||
acc += a*srow[g];
|
||||
}
|
||||
y[(int64_t)s*O+o]=acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- IDOT: integer dot kernels (int8-quantized activations) --------------- */
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
#define IDOT_KERNEL "avx512-vnni"
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
#define IDOT_KERNEL "avx-vnni"
|
||||
#elif defined(__AVX2__)
|
||||
#define IDOT_KERNEL "avx2"
|
||||
#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
#define IDOT_KERNEL "neon-i8mm"
|
||||
#elif defined(__ARM_NEON)
|
||||
#define IDOT_KERNEL "neon"
|
||||
#elif defined(__VSX__)
|
||||
#define IDOT_KERNEL "vsx"
|
||||
#else
|
||||
#define IDOT_KERNEL "scalar"
|
||||
#endif
|
||||
static int g_idot=1;
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
|
||||
static int g_i4s=1;
|
||||
#elif defined(__VSX__)
|
||||
static int g_i4s=1;
|
||||
#else
|
||||
static int g_i4s=2;
|
||||
#endif
|
||||
|
||||
static inline float qrow_i8(const float *x, int8_t *q, int I){
|
||||
float amax=0; for(int i=0;i<I;i++){ float a=fabsf(x[i]); if(a>amax)amax=a; }
|
||||
float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s;
|
||||
for(int i=0;i<I;i++) q[i]=(int8_t)lrintf(x[i]*inv);
|
||||
return s;
|
||||
}
|
||||
|
||||
/* dot int8*int8 */
|
||||
static inline int32_t dot_i8i8(const int8_t *w, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
__m512i acc=_mm512_setzero_si512();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m512i wv=_mm512_loadu_si512((const void*)(w+i));
|
||||
__m512i xv=_mm512_loadu_si512((const void*)(x+i));
|
||||
__mmask64 neg=_mm512_movepi8_mask(wv);
|
||||
__m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
|
||||
acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=_mm512_reduce_add_epi32(acc);
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
/* 4 accumulatori indipendenti (64 byte/iter): un solo acc incatena i vpdpbusd
|
||||
* (latenza-bound ~5c). Somme intere associative -> bit-identico. Stessa struttura
|
||||
* dei 4 accumulatori del ramo NEON piu' sotto.
|
||||
* EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
|
||||
* adds are associative, so the result is bit-identical (mirrors the NEON path). */
|
||||
__m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m128i w0=_mm_loadu_si128((const __m128i*)(w+i)), x0=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i w1=_mm_loadu_si128((const __m128i*)(w+i+16)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
|
||||
__m128i w2=_mm_loadu_si128((const __m128i*)(w+i+32)), x2=_mm_loadu_si128((const __m128i*)(x+i+32));
|
||||
__m128i w3=_mm_loadu_si128((const __m128i*)(w+i+48)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
|
||||
a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
|
||||
a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
|
||||
a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
|
||||
a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
|
||||
}
|
||||
__m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
|
||||
for(;i+16<=I;i+=16){
|
||||
__m128i wv=_mm_loadu_si128((const __m128i*)(w+i));
|
||||
__m128i xv=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),_mm_sign_epi8(xv,wv));
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
#elif defined(__AVX2__)
|
||||
__m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1);
|
||||
for(;i+32<=I;i+=32){
|
||||
__m256i wv=_mm256_loadu_si256((const __m256i*)(w+i));
|
||||
__m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
|
||||
__m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
|
||||
acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
|
||||
}
|
||||
sum=hsum256_i32(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
#if defined(__ARM_FEATURE_DOTPROD)
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
|
||||
for(;i+64<=I;i+=64){
|
||||
a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i));
|
||||
a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16));
|
||||
a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32));
|
||||
a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i));
|
||||
sum=vaddvq_s32(acc);
|
||||
#else
|
||||
int32x4_t acc=vdupq_n_s32(0);
|
||||
for(;i+16<=I;i+=16){
|
||||
int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i);
|
||||
int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv));
|
||||
p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#endif
|
||||
#elif defined(__VSX__)
|
||||
__vector signed int acc=vec_splats(0);
|
||||
const __vector signed char vz=vec_splats((signed char)0);
|
||||
for(;i+16<=I;i+=16){
|
||||
__vector signed char wv=vec_xl(0,(const signed char*)(w+i));
|
||||
__vector signed char xv=vec_xl(0,(const signed char*)(x+i));
|
||||
__vector __bool char neg=vec_cmplt(wv,vz);
|
||||
__vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg);
|
||||
__vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg);
|
||||
acc=vec_msum(xs,wa,acc);
|
||||
}
|
||||
sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
|
||||
#endif
|
||||
for(;i<I;i++) sum+=(int32_t)w[i]*x[i];
|
||||
return sum;
|
||||
}
|
||||
|
||||
/* dot int4(packed)*int8 */
|
||||
static inline int32_t dot_i4i8(const uint8_t *w4, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
const __m256i m4v=_mm256_set1_epi8(0x0F);
|
||||
const __m512i b8v=_mm512_set1_epi8(8);
|
||||
const __m512i xidx=_mm512_setr_epi64(0,1,4,5,2,3,6,7);
|
||||
__m512i acc=_mm512_setzero_si512();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m256i by=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
|
||||
__m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v);
|
||||
__m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi);
|
||||
__m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v);
|
||||
__m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i)));
|
||||
__mmask64 neg=_mm512_movepi8_mask(wv);
|
||||
__m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
|
||||
acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=_mm512_reduce_add_epi32(acc);
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
/* 4 accumulatori indipendenti (64 elementi = 32 byte packed/iter): un solo acc
|
||||
* incatena i vpdpbusd (latenza-bound ~5c). Somme intere associative -> bit-identico.
|
||||
* Stessa struttura dei 4 accumulatori del ramo NEON piu' sotto.
|
||||
* EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
|
||||
* adds are associative, so the result is bit-identical (mirrors the NEON path). */
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8);
|
||||
__m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m128i by0=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); /* elem i..i+31 */
|
||||
__m128i by1=_mm_loadu_si128((const __m128i*)(w4+(i>>1)+16)); /* elem i+32..i+63 */
|
||||
__m128i lo0=_mm_and_si128(by0,m4), hi0=_mm_and_si128(_mm_srli_epi16(by0,4),m4);
|
||||
__m128i lo1=_mm_and_si128(by1,m4), hi1=_mm_and_si128(_mm_srli_epi16(by1,4),m4);
|
||||
__m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo0,hi0),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo0,hi0),b8);
|
||||
__m128i w2=_mm_sub_epi8(_mm_unpacklo_epi8(lo1,hi1),b8), w3=_mm_sub_epi8(_mm_unpackhi_epi8(lo1,hi1),b8);
|
||||
__m128i x0=_mm_loadu_si128((const __m128i*)(x+i)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
|
||||
__m128i x2=_mm_loadu_si128((const __m128i*)(x+i+32)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
|
||||
a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
|
||||
a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
|
||||
a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
|
||||
a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
|
||||
}
|
||||
__m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
|
||||
for(;i+32<=I;i+=32){ /* 32-nibble remainder: 2 dpbusd, same unpack */
|
||||
__m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo,hi),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo,hi),b8);
|
||||
__m128i x0=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
#elif defined(__AVX2__)
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8);
|
||||
const __m256i ones=_mm256_set1_epi16(1);
|
||||
__m256i acc=_mm256_setzero_si256();
|
||||
for(;i+32<=I;i+=32){
|
||||
__m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8);
|
||||
__m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
|
||||
__m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
|
||||
acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
|
||||
}
|
||||
sum=hsum256_i32(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
|
||||
#if defined(__ARM_FEATURE_DOTPROD)
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
|
||||
for(;i+64<=I;i+=64){
|
||||
uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16);
|
||||
uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4));
|
||||
uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4));
|
||||
a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i));
|
||||
a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16));
|
||||
a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32));
|
||||
a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t by=vld1q_u8(w4+(i>>1));
|
||||
uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
|
||||
acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i));
|
||||
acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16));
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#else
|
||||
int32x4_t acc=vdupq_n_s32(0);
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t by=vld1q_u8(w4+(i>>1));
|
||||
uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
|
||||
int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q);
|
||||
int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q);
|
||||
int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16);
|
||||
int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0));
|
||||
p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1));
|
||||
p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#endif
|
||||
#elif defined(__VSX__)
|
||||
const __vector unsigned char m4v=vec_splats((unsigned char)0x0F);
|
||||
const __vector unsigned char sh4=vec_splats((unsigned char)4);
|
||||
const __vector signed char b8v=vec_splats((signed char)8);
|
||||
const __vector signed char vz=vec_splats((signed char)0);
|
||||
__vector signed int acc=vec_splats(0);
|
||||
for(;i+32<=I;i+=32){
|
||||
__vector unsigned char by=vec_xl(0,w4+(i>>1));
|
||||
__vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4);
|
||||
__vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v);
|
||||
__vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v);
|
||||
__vector signed char x0=vec_xl(0,(const signed char*)(x+i));
|
||||
__vector signed char x1=vec_xl(0,(const signed char*)(x+i+16));
|
||||
__vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz);
|
||||
acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0),
|
||||
(__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc);
|
||||
acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1),
|
||||
(__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc);
|
||||
}
|
||||
sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t b=w4[i>>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; }
|
||||
if(i<I){ uint8_t b=w4[i>>1]; sum+=((int)(b&0xF)-8)*x[i]; }
|
||||
return sum;
|
||||
}
|
||||
|
||||
/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1,
|
||||
int8x16_t xs, int8x16_t xs1){
|
||||
acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)),
|
||||
vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1)));
|
||||
return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)),
|
||||
vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1)));
|
||||
}
|
||||
static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q,
|
||||
const float *scale, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<(O&~1);o+=2){
|
||||
const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I;
|
||||
float sc0=scale[o], sc1=scale[o+1];
|
||||
for(int s=0;s<(S&~1);s+=2){
|
||||
const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
|
||||
for(;i+64<=I;i+=64){
|
||||
a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16));
|
||||
a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32));
|
||||
a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48));
|
||||
}
|
||||
for(;i+16<=I;i+=16)
|
||||
a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i));
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
|
||||
int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
|
||||
for(;i<I;i++){ int a=wo[i],b=wo1[i],u=xs[i],v=xs1[i];
|
||||
d00+=a*u; d01+=a*v; d10+=b*u; d11+=b*v; }
|
||||
y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
|
||||
y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
|
||||
y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
|
||||
}
|
||||
if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
|
||||
y[(int64_t)s*O+o] =(float)dot_i8i8(wo, xs,I)*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)]=(float)dot_i8i8(wo1,xs,I)*sc1*sx[s]; }
|
||||
}
|
||||
if(O&1){ int o=O-1; const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i8i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
static void matmul_i4_idot_mm(float *y, const int8_t *xq, const float *sx, const uint8_t *q4,
|
||||
const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<(O&~1);o+=2){
|
||||
const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
|
||||
const uint8_t *wo=q4+(int64_t)o*rb, *wo1=q4+(int64_t)(o+1)*rb;
|
||||
float sc0=scale[o], sc1=scale[o+1];
|
||||
for(int s=0;s<(S&~1);s+=2){
|
||||
const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
|
||||
for(;i+64<=I;i+=64){
|
||||
uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
|
||||
uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16);
|
||||
uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
|
||||
uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
|
||||
uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4));
|
||||
uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4));
|
||||
a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
|
||||
vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
|
||||
a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q),
|
||||
vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32));
|
||||
a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48));
|
||||
}
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
|
||||
uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
|
||||
uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
|
||||
a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
|
||||
vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
|
||||
int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
|
||||
for(;i+1<I;i+=2){ uint8_t bo=wo[i>>1], bo1=wo1[i>>1];
|
||||
int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8;
|
||||
int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1];
|
||||
d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; }
|
||||
if(i<I){ uint8_t bo=wo[i>>1], bo1=wo1[i>>1];
|
||||
int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8;
|
||||
d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; }
|
||||
y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
|
||||
y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
|
||||
y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
|
||||
}
|
||||
if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
|
||||
y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; }
|
||||
}
|
||||
if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i4i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
#endif
|
||||
|
||||
/* ---- IDOT dispatch (int8-quantized activations) --------------------------- */
|
||||
static void matmul_q_idot(float *y, const int8_t *xq, const float *sx, const int8_t *q,
|
||||
const float *scale, int S, int I, int O){
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
if(S>=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; }
|
||||
#endif
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i8i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
static void matmul_i4_idot(float *y, const int8_t *xq, const float *sx, const uint8_t *q4,
|
||||
const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
if(S>=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; }
|
||||
#endif
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i4i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
|
||||
/* ---- per-thread quantization scratch -------------------------------------- */
|
||||
typedef struct { int8_t *xq; size_t xq_cap; float *sx; size_t sx_cap; } QScratch;
|
||||
static _Thread_local QScratch g_qscratch;
|
||||
static void quant_scratch(size_t xn, size_t sn, int8_t **xq, float **sx){
|
||||
if(xn>g_qscratch.xq_cap){
|
||||
int8_t *p=realloc(g_qscratch.xq,xn);
|
||||
if(!p){ fprintf(stderr,"OOM quant scratch\n"); exit(1); }
|
||||
g_qscratch.xq=p; g_qscratch.xq_cap=xn;
|
||||
}
|
||||
if(sn>g_qscratch.sx_cap){
|
||||
float *p=realloc(g_qscratch.sx,sn*sizeof(float));
|
||||
if(!p){ fprintf(stderr,"OOM quant scales\n"); exit(1); }
|
||||
g_qscratch.sx=p; g_qscratch.sx_cap=sn;
|
||||
}
|
||||
*xq=g_qscratch.xq; *sx=g_qscratch.sx;
|
||||
}
|
||||
|
||||
/* ---- f32 -> quantized packing --------------------------------------------- */
|
||||
static void quantize_rows(const float *w, int8_t *q, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
int8_t *qr=q+(int64_t)o*I;
|
||||
for(int i=0;i<I;i++){ int v=(int)lrintf(wr[i]/s); if(v>qmax)v=qmax; if(v<-qmax-1)v=-qmax-1; qr[i]=(int8_t)v; }
|
||||
}
|
||||
}
|
||||
static void pack_int4(const float *w, uint8_t *q4, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1, rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
uint8_t *qr=q4+(int64_t)o*rb;
|
||||
for(int i=0;i<I;i+=2){
|
||||
int v0=(int)lrintf(wr[i]/s); if(v0>qmax)v0=qmax; if(v0<-8)v0=-8;
|
||||
int v1=0; if(i+1<I){ v1=(int)lrintf(wr[i+1]/s); if(v1>qmax)v1=qmax; if(v1<-8)v1=-8; }
|
||||
qr[i>>1] = (uint8_t)((v0+8) | ((v1+8)<<4));
|
||||
}
|
||||
}
|
||||
}
|
||||
/* quantize w[O,I] f32 -> int3-g64 (fmt=5): per 64-input group, symmetric absmax
|
||||
* (qmax=3, clamp [-4,3], stored v+4), 16B low plane + 8B high plane, ONE f32 scale
|
||||
* per group. Same math as tools/quant_ablation.py `_quant_last_dim(bits=3, group=64)`
|
||||
* (#132), here with real bit packing. */
|
||||
static void pack_int3_g64(const float *w, uint8_t *q3, float *scale, int O, int I){
|
||||
int64_t ng=i3_groups(I), rb=i3_rowbytes(I);
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){
|
||||
const float *wr=w+(int64_t)o*I;
|
||||
uint8_t *qr=q3+(int64_t)o*rb;
|
||||
float *sr=scale+(int64_t)o*ng;
|
||||
for(int64_t g=0; g<ng; g++){
|
||||
int base=(int)(g*I3_GROUP), n = I-base < I3_GROUP ? I-base : I3_GROUP;
|
||||
float amax=0;
|
||||
for(int k=0;k<n;k++){ float a=fabsf(wr[base+k]); if(a>amax)amax=a; }
|
||||
float s=amax/3.f; if(s<1e-8f)s=1e-8f; sr[g]=s;
|
||||
uint8_t *lo=qr+g*I3_GBYTES, *hi=lo+16;
|
||||
memset(lo,0,I3_GBYTES);
|
||||
for(int k=0;k<n;k++){
|
||||
int v=(int)lrintf(wr[base+k]/s); if(v>3)v=3; if(v<-4)v=-4;
|
||||
unsigned u=(unsigned)(v+4); /* 0..7 */
|
||||
lo[k>>2] |= (uint8_t)((u&3)<<((k&3)*2));
|
||||
hi[k>>3] |= (uint8_t)(((u>>2)&1)<<(k&7));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void pack_int2(const float *w, uint8_t *q2, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1, rb=(I+3)/4;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
uint8_t *qr=q2+(int64_t)o*rb;
|
||||
for(int i=0;i<I;i+=4){ uint8_t byte=0;
|
||||
for(int k=0;k<4 && i+k<I;k++){ int v=(int)lrintf(wr[i+k]/s); if(v>qmax)v=qmax; if(v<-2)v=-2; byte|=(uint8_t)((v+2)<<(k*2)); }
|
||||
qr[i>>2]=byte;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif /* COLI_QUANT_H */
|
||||
@@ -0,0 +1,28 @@
|
||||
{
|
||||
"prompt_ids": [
|
||||
510,
|
||||
5347,
|
||||
273,
|
||||
6181,
|
||||
310
|
||||
],
|
||||
"full_ids": [
|
||||
510,
|
||||
5347,
|
||||
273,
|
||||
6181,
|
||||
310,
|
||||
7785,
|
||||
15,
|
||||
187,
|
||||
187,
|
||||
510,
|
||||
3565,
|
||||
3448,
|
||||
273,
|
||||
6181,
|
||||
310,
|
||||
5112,
|
||||
15
|
||||
]
|
||||
}
|
||||
+190
-23
@@ -166,14 +166,33 @@ def discover_gpus():
|
||||
return devices
|
||||
|
||||
|
||||
def _physical_cores_warn(message):
|
||||
"""Visibility for a mis-detected core count: a silent "1" here becomes
|
||||
OMP_NUM_THREADS=1 and pins the whole run to a single core (#325). Emit on
|
||||
stderr so it surfaces in the [PLAN]/[OMP] stream without being swallowed."""
|
||||
print(f"[plan] warning: {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def physical_cpu_count():
|
||||
"""Number of physical CPU cores (not SMT siblings).
|
||||
|
||||
Per-expert matmul regions are tiny and back-to-back; two SMT siblings share
|
||||
one AVX-512 unit and contend, so logical (SMT) counts over-subscribe and
|
||||
hurt throughput. We want true physical cores. A silent 1 here propagates to
|
||||
OMP_NUM_THREADS=1 and pins the run to one core (#325), so every fallback
|
||||
must be visible, never just ``or 1``.
|
||||
"""
|
||||
if sys.platform == "win32":
|
||||
# os.cpu_count() conta i processori logici (SMT): 2 thread/core saturano
|
||||
# le unita' AVX-512 e peggiorano il matmul. Contiamo i core fisici veri
|
||||
# con GetLogicalProcessorInformationEx(RelationProcessorCore).
|
||||
# Contiamo i core fisici veri con GetLogicalProcessorInformationEx
|
||||
# (RelationProcessorCore). Le firme vanno dichiarate: su Python a 64 bit
|
||||
# una WinAPI non dichiarata ritorna c_int (32 bit) e riceve i puntatori
|
||||
# come c_int di default, quindi il probe puo' fallire silenziosamente.
|
||||
try:
|
||||
import ctypes
|
||||
k32 = ctypes.windll.kernel32
|
||||
k32.GetLogicalProcessorInformationEx.argtypes = [
|
||||
ctypes.c_uint, ctypes.c_void_p, ctypes.POINTER(ctypes.c_ulong)]
|
||||
k32.GetLogicalProcessorInformationEx.restype = ctypes.c_int
|
||||
need = ctypes.c_ulong(0)
|
||||
k32.GetLogicalProcessorInformationEx(0, None, ctypes.byref(need))
|
||||
buf = (ctypes.c_char * need.value)()
|
||||
@@ -189,18 +208,69 @@ def physical_cpu_count():
|
||||
off += size
|
||||
if cores:
|
||||
return cores
|
||||
except (OSError, ValueError, AttributeError):
|
||||
pass
|
||||
_physical_cores_warn("GetLogicalProcessorInformationEx returned no cores")
|
||||
except (OSError, ValueError, AttributeError) as error:
|
||||
_physical_cores_warn(f"Windows core probe failed: {error}")
|
||||
try:
|
||||
# Ask lscpu for exactly core,socket and dedupe on (core, socket).
|
||||
# Counting un-deduplicated rows would return logical threads (SMT),
|
||||
# which was the original over-subscription bug. Empty fields ("-")
|
||||
# mark an offline core/socket and fail int() -> skipped.
|
||||
#
|
||||
# Column layout robustness: `lscpu -p=<list>` emits *exactly* the
|
||||
# requested columns (no CPU prefix), while bare `lscpu -p` prepends
|
||||
# CPU. We requested two columns, but take the LAST TWO fields so the
|
||||
# parser stays correct whether or not a CPU column is present
|
||||
# (JustVugg review: the previous fields[1]/fields[2] indexing assumed
|
||||
# a 3-column layout and regressed 2-column output to the logical
|
||||
# count -- the opposite of the fix).
|
||||
result = subprocess.run(["lscpu", "-p=core,socket"], text=True,
|
||||
capture_output=True, check=True, timeout=5)
|
||||
cores = {tuple(map(int, line.split(","))) for line in result.stdout.splitlines()
|
||||
if line and not line.startswith("#")}
|
||||
cores = set()
|
||||
for line in result.stdout.splitlines():
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
fields = line.split(",")
|
||||
if len(fields) < 2:
|
||||
continue
|
||||
try:
|
||||
core, socket = int(fields[-2]), int(fields[-1])
|
||||
except ValueError:
|
||||
continue # "-" for an offline core/socket
|
||||
cores.add((core, socket))
|
||||
if cores:
|
||||
return len(cores)
|
||||
except (OSError, ValueError, subprocess.SubprocessError):
|
||||
pass
|
||||
return os.cpu_count() or 1
|
||||
except (OSError, ValueError, subprocess.SubprocessError) as error:
|
||||
_physical_cores_warn(f"lscpu core probe failed: {error}")
|
||||
logical = os.cpu_count()
|
||||
if not logical:
|
||||
_physical_cores_warn(
|
||||
"could not detect any CPU cores; falling back to 1. "
|
||||
"Set OMP_NUM_THREADS manually to fix single-core decode (#325).")
|
||||
return 1
|
||||
_physical_cores_warn(
|
||||
f"physical-core probes unavailable; using {logical} logical CPUs "
|
||||
f"(SMT may over-subscribe). Set OMP_NUM_THREADS to physical cores if slow.")
|
||||
return logical
|
||||
|
||||
|
||||
def _resolve_physical_cores(physical_cpus):
|
||||
"""Coerce the build_plan() physical-core argument to a sane positive int.
|
||||
|
||||
A None/0/None-ish value reaching here means physical_cpu_count() already
|
||||
warned; clamp to 1 (so the engine always gets a positive team size) but keep
|
||||
that clamp visible rather than silently masking it as the old ``max(1, int())``
|
||||
did (#325)."""
|
||||
try:
|
||||
count = int(physical_cpus or 0)
|
||||
except (TypeError, ValueError):
|
||||
count = 0
|
||||
if count < 1:
|
||||
_physical_cores_warn(
|
||||
"physical core count resolved to 0; defaulting to 1. "
|
||||
"Set OMP_NUM_THREADS to fix single-core decode (#325).")
|
||||
return 1
|
||||
return count
|
||||
|
||||
|
||||
def cpu_socket_count():
|
||||
@@ -219,6 +289,63 @@ def cpu_socket_count():
|
||||
return 1
|
||||
|
||||
|
||||
def _auto_tune(bottleneck_class, projected_hit, gpus, cpu_sockets, plan_has_metal):
|
||||
"""Derive tuning knobs from the bottleneck classification."""
|
||||
tune = {}
|
||||
has_gpu = bool(gpus)
|
||||
n_gpu = len(gpus)
|
||||
|
||||
# MTP: costs more than it saves when compute-bound (#389 measured 42% loss)
|
||||
# or streaming-bound (#467 measured 32% loss under CUDA at 85% hit).
|
||||
# EXCEPTION: an explicit COLI_CUDA_MTP=1 in the environment is a documented
|
||||
# opt-in to test speculation under CUDA (glm.c resolves DRAFT=-1 -> 3 only
|
||||
# when it sees the var). Exporting DRAFT=0 here preempted that auto path,
|
||||
# so the opt-in was silently inert on the Windows bare-run/auto-tier flows
|
||||
# (#467): respect it and let the engine's auto path take over. Unset still
|
||||
# gets DRAFT=0 -> MTP off, which is the measured-correct default.
|
||||
if os.environ.get("COLI_CUDA_MTP") == "1":
|
||||
pass # explicit opt-in: leave DRAFT to the engine's auto resolution
|
||||
elif bottleneck_class == "compute":
|
||||
tune["DRAFT"] = {"value": "0",
|
||||
"reason": "compute-bound: MTP batch overhead exceeds yield"}
|
||||
elif bottleneck_class == "disk" and projected_hit < 0.90:
|
||||
tune["DRAFT"] = {"value": "0",
|
||||
"reason": "low hit rate: MTP widens expert union, adds disk reads"}
|
||||
# otherwise leave DRAFT unset (engine default: auto)
|
||||
|
||||
# PIPE: resident pipeline mode depends on GPU count
|
||||
if has_gpu and n_gpu == 1:
|
||||
tune["COLI_CUDA_PIPE"] = {"value": "1",
|
||||
"reason": "single GPU: S=1 pipeline gate"}
|
||||
elif has_gpu and n_gpu > 1:
|
||||
tune["COLI_CUDA_PIPE"] = {"value": "2",
|
||||
"reason": "multi-GPU: residual stays on-device across layers"}
|
||||
elif not has_gpu and bottleneck_class == "disk":
|
||||
tune["PIPE"] = {"value": "1",
|
||||
"reason": "overlap disk reads with resident expert compute"}
|
||||
|
||||
# NUMA: selective interleave for GPU hosts, blanket hint for CPU-only
|
||||
if cpu_sockets > 1 and has_gpu:
|
||||
tune["COLI_NUMA"] = {"value": "1",
|
||||
"reason": "multi-socket + GPU: interleave expert slabs, protect DMA buffers"}
|
||||
elif cpu_sockets > 1 and not has_gpu:
|
||||
tune["COLI_NUMA"] = {"value": "1",
|
||||
"reason": "multi-socket CPU-only: interleave expert slabs across nodes"}
|
||||
tune["_numa_hint"] = "numactl --interleave=all may perform better on CPU-only hosts"
|
||||
|
||||
# OMP: kill hot-thread spin when GPU/Metal owns the power budget
|
||||
if plan_has_metal:
|
||||
tune["COLI_NO_OMP_TUNE"] = {"value": "1",
|
||||
"reason": "Metal: OMP spin-wait steals GPU power budget"}
|
||||
|
||||
# PIN: fully resident if RAM allows and no GPU tier competes
|
||||
if projected_hit >= 0.99 and not has_gpu:
|
||||
tune["PIN_GB"] = {"value": "all",
|
||||
"reason": "enough RAM for full expert residency"}
|
||||
|
||||
return tune
|
||||
|
||||
|
||||
POLICIES = {
|
||||
"quality": {"preserve_quantization": True, "preserve_router": True},
|
||||
"balanced": {"preserve_quantization": True, "preserve_router": True},
|
||||
@@ -290,19 +417,35 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
|
||||
if cold_bytes:
|
||||
warnings.append("cold expert misses may reach disk; normal decode speed depends on hit rate")
|
||||
|
||||
total_expert = info["expert_bytes"]
|
||||
resident_expert = hot_bytes + warm_bytes
|
||||
projected_hit = resident_expert / total_expert if total_expert else 1.0
|
||||
|
||||
if cold_bytes:
|
||||
bottleneck = "disk expert misses"
|
||||
elif warm_bytes:
|
||||
bottleneck = "CPU expert compute and RAM bandwidth"
|
||||
bottleneck_class = "disk"
|
||||
elif warm_bytes and gpus:
|
||||
bottleneck = "CPU expert tail and GPU compute"
|
||||
bottleneck_class = "mixed"
|
||||
elif projected_hit >= 0.99:
|
||||
if gpus:
|
||||
bottleneck = "GPU compute and interconnect"
|
||||
else:
|
||||
bottleneck = "CPU expert compute (fully resident)"
|
||||
bottleneck_class = "compute"
|
||||
else:
|
||||
bottleneck = "GPU compute and interconnect"
|
||||
bottleneck = "CPU expert compute and RAM bandwidth"
|
||||
bottleneck_class = "memory"
|
||||
|
||||
tune = _auto_tune(bottleneck_class, projected_hit, gpus, cpu_sockets,
|
||||
plan_has_metal=False)
|
||||
|
||||
return {
|
||||
"version": 2,
|
||||
"policy": {"name": policy, **POLICIES[policy],
|
||||
"quality_preserving": policy != "experimental-fast"},
|
||||
"model": {key: value for key, value in info.items() if key != "config"},
|
||||
"cpu": {"physical_cores": max(1, int(physical_cpus)),
|
||||
"cpu": {"physical_cores": _resolve_physical_cores(physical_cpus),
|
||||
"sockets": max(1, int(cpu_sockets)),
|
||||
"thread_policy": "physical-cores"},
|
||||
"tiers": {
|
||||
@@ -317,6 +460,9 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
|
||||
"expert_capacity": vram_experts, "requires_host_backing": False},
|
||||
},
|
||||
"expected_bottleneck": bottleneck,
|
||||
"bottleneck_class": bottleneck_class,
|
||||
"projected_hit_rate": round(projected_hit, 4),
|
||||
"tune": tune,
|
||||
"decisions": [
|
||||
{"target": "VRAM", "reason": "profile-ranked hot experts"},
|
||||
{"target": "RAM", "reason": "warm experts execute on CPU without quality loss"},
|
||||
@@ -331,15 +477,23 @@ def environment_for_plan(plan, env=None, cuda_enabled=True):
|
||||
result = dict(env or {})
|
||||
result.setdefault("COLI_POLICY", plan["policy"]["name"])
|
||||
result.setdefault("OMP_NUM_THREADS", str(plan["cpu"]["physical_cores"]))
|
||||
if sys.platform != "win32":
|
||||
# la libgomp di MinGW non supporta l'affinity su Windows
|
||||
# ("Affinity not supported on this configuration"): non impostarle li'.
|
||||
result.setdefault("OMP_PROC_BIND", "spread")
|
||||
result.setdefault("OMP_PLACES", "cores")
|
||||
if sys.platform.startswith("linux") and plan["cpu"].get("sockets", 1) > 1:
|
||||
# Selectively interleave large expert/dense slabs across memory controllers.
|
||||
# Unlike blanket numactl interleave, this leaves CUDA staging buffers local.
|
||||
result.setdefault("COLI_NUMA", "1")
|
||||
# NOTE: we intentionally do NOT set OMP_PROC_BIND / OMP_PLACES here.
|
||||
# The engine's own hot-thread tuning (glm.c main(), the COLI_OMP_TUNED
|
||||
# self-exec) sets OMP_PROC_BIND=close with overwrite=0 -- it prefers
|
||||
# packing the team onto adjacent cores for the tiny back-to-back per-expert
|
||||
# matmuls. Pre-setting OMP_PROC_BIND=spread here ran first and won (the
|
||||
# engine's overwrite=0 setenv could not override an already-set var), and
|
||||
# spread + OMP_PLACES=cores collapsed the team to one CPU on some libgomp /
|
||||
# multi-socket topologies (#325: --auto-tier pinned decode to 1 core on a
|
||||
# 64-core box even with OMP_NUM_THREADS=64). Leaving affinity to the engine
|
||||
# makes --auto-tier match the plain (working) path. A user who wants a
|
||||
# specific policy can still set OMP_PROC_BIND/OMP_PLACES in the environment
|
||||
# themselves -- setdefault above only covers OMP_NUM_THREADS.
|
||||
tune = plan.get("tune", {})
|
||||
for key, entry in tune.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
result.setdefault(key, entry["value"])
|
||||
if plan["policy"]["name"] == "balanced":
|
||||
result.setdefault("REPIN", "64")
|
||||
ram = plan["tiers"]["ram"]
|
||||
@@ -386,5 +540,18 @@ def format_plan(plan):
|
||||
else:
|
||||
lines.append("VRAM no NVIDIA device detected · CPU path")
|
||||
lines.append(f"limit {plan['expected_bottleneck']}")
|
||||
hit = plan.get("projected_hit_rate", 0)
|
||||
lines.append(f"hit {hit:.0%} projected expert residency")
|
||||
tune = plan.get("tune", {})
|
||||
if tune:
|
||||
lines.append("")
|
||||
lines.append("auto-tune:")
|
||||
for key, entry in tune.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
lines.append(f" {key}={entry['value']:12s} {entry['reason']}")
|
||||
hint = tune.get("_numa_hint")
|
||||
if hint:
|
||||
lines.append(f" hint: {hint}")
|
||||
lines.extend(f"warn {warning}" for warning in plan["warnings"])
|
||||
return "\n".join(lines)
|
||||
|
||||
+153
@@ -0,0 +1,153 @@
|
||||
/* sample.h — sampling (temperature + nucleus) and stop-set management.
|
||||
* Header-only: all functions are static — include from the main engine file. */
|
||||
#ifndef SAMPLE_H
|
||||
#define SAMPLE_H
|
||||
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include "tok.h"
|
||||
|
||||
/* ---- RNG (xorshift64*) -------------------------------------------------- */
|
||||
static uint64_t g_rng = 0x9E3779B97F4A7C15ULL;
|
||||
static inline double rndu(void){
|
||||
g_rng ^= g_rng << 13; g_rng ^= g_rng >> 7; g_rng ^= g_rng << 17;
|
||||
return (double)(g_rng >> 11) * (1.0 / 9007199254740992.0);
|
||||
}
|
||||
|
||||
/* ---- argmax over a float vector ----------------------------------------- */
|
||||
static inline int argmax_v(const float *lo, int V){
|
||||
int b=-1; float bv=-INFINITY;
|
||||
for(int i=0;i<V;i++){ float x=lo[i]; if(x==x && x>bv){ bv=x; b=i; } }
|
||||
return b<0?0:b;
|
||||
}
|
||||
|
||||
/* ---- distribution buffers (reused, single-threaded decode) --------------- */
|
||||
static float *g_pbuf = NULL;
|
||||
static int *g_pidx = NULL;
|
||||
|
||||
/* sift-down on max-heap in h[0..n), key = g_pbuf[h[i]] (#335: partial top-p).
|
||||
* "hole" variant: carries the root value and deposits only at the end, so
|
||||
* heapify is O(V) and each pop is O(log n) without qsort on the full vocab. */
|
||||
static void topp_siftdown(int *h, int n, int i){
|
||||
int iv = h[i]; float kv = g_pbuf[iv];
|
||||
for (;;) {
|
||||
int l = 2*i + 1;
|
||||
if (l >= n) break;
|
||||
int b = l; if (l+1 < n && g_pbuf[h[l+1]] > g_pbuf[h[l]]) b = l+1;
|
||||
if (g_pbuf[h[b]] <= kv) break;
|
||||
h[i] = h[b]; i = b;
|
||||
}
|
||||
h[i] = iv;
|
||||
}
|
||||
|
||||
/* build the target distribution in g_pbuf: softmax(lo/temp) truncated to
|
||||
* top-p g_nuc. Invariant: g_pbuf stays indexed by token-id (never reordered);
|
||||
* the truncated tail is zeroed (dist_sample reads by id directly).
|
||||
* Requires: g_temp, g_nuc, falloc() — declared in the main engine file. */
|
||||
static void dist_build(const float *lo, int V){
|
||||
if (!g_pbuf) { g_pbuf = falloc(V); g_pidx = malloc(V * sizeof(int)); }
|
||||
int mxi = -1; float mx = 0;
|
||||
for (int i = 0; i < V; i++)
|
||||
if (isfinite(lo[i]) && (mxi < 0 || lo[i] > mx)) { mx = lo[i]; mxi = i; }
|
||||
double s = 0; float invt = 1.f / (g_temp > 1e-4f ? g_temp : 1e-4f);
|
||||
if (mxi >= 0) {
|
||||
for (int i = 0; i < V; i++) {
|
||||
g_pbuf[i] = isfinite(lo[i]) ? expf((lo[i] - mx) * invt) : 0.f;
|
||||
s += g_pbuf[i];
|
||||
}
|
||||
}
|
||||
if (mxi < 0 || !isfinite(s) || s <= 0.0) {
|
||||
static int warned = 0;
|
||||
if (!warned) { warned = 1; fprintf(stderr,
|
||||
"[SAMPLE] warning: non-finite logits (NaN/Inf) — falling back to argmax; "
|
||||
"output may be degraded. This usually means a numerical blow-up upstream.\n"); }
|
||||
int a = (mxi >= 0) ? mxi : 0;
|
||||
for (int i = 0; i < V; i++) g_pbuf[i] = 0.f;
|
||||
g_pbuf[a] = 1.f;
|
||||
return;
|
||||
}
|
||||
for (int i = 0; i < V; i++) g_pbuf[i] /= (float)s;
|
||||
if (g_nuc > 0 && g_nuc < 1.f) {
|
||||
for (int i = 0; i < V; i++) g_pidx[i] = i;
|
||||
for (int i = V/2-1; i >= 0; i--) topp_siftdown(g_pidx, V, i);
|
||||
double s2 = 0, cum = 0; int out = V;
|
||||
do {
|
||||
int root = g_pidx[0];
|
||||
g_pidx[0] = g_pidx[--out]; g_pidx[out] = root;
|
||||
s2 += g_pbuf[root]; cum += g_pbuf[root];
|
||||
if (out > 0) topp_siftdown(g_pidx, out, 0);
|
||||
} while (cum < g_nuc && out > 0);
|
||||
for (int i = 0; i < out; i++) g_pbuf[g_pidx[i]] = 0;
|
||||
float s2f = (float)s2;
|
||||
for (int i = out; i < V; i++) g_pbuf[g_pidx[i]] /= s2f;
|
||||
}
|
||||
}
|
||||
|
||||
/* sample from g_pbuf; ban>=0 excludes that token (renormalizing on the fly) */
|
||||
static int dist_sample(int V, int ban){
|
||||
double z = 1.0 - (ban >= 0 ? g_pbuf[ban] : 0.0);
|
||||
if (z <= 1e-12) z = 1e-12;
|
||||
double u = rndu() * z, cum = 0;
|
||||
for (int i = 0; i < V; i++) { if (i == ban) continue; cum += g_pbuf[i]; if (cum >= u) return i; }
|
||||
for (int i = V-1; i >= 0; i--) if (i != ban && g_pbuf[i] > 0) return i;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* next token from logits: greedy if g_temp<=0, sampling otherwise.
|
||||
* ban = token excluded because it was rejected by speculative verification. */
|
||||
static int pick_tok(const float *lo, int V, int ban){
|
||||
if (g_temp <= 0) return argmax_v(lo, V);
|
||||
dist_build(lo, V);
|
||||
return dist_sample(V, ban);
|
||||
}
|
||||
|
||||
/* ---- stop set ----------------------------------------------------------- */
|
||||
static int g_stop[64], g_nstop = 0;
|
||||
static inline int is_stop(int t){
|
||||
for (int i = 0; i < g_nstop; i++) if (t == g_stop[i]) return 1;
|
||||
return 0;
|
||||
}
|
||||
/* T=NULL -> config stops only (validation/oracle, where the tokenizer is not needed). */
|
||||
static void stops_arm_tok(const Cfg *c, int tok_eos, Tok *T){
|
||||
g_nstop = 0;
|
||||
for (int i = 0; i < c->n_stop && g_nstop < 64; i++) g_stop[g_nstop++] = c->stop_ids[i];
|
||||
if (tok_eos >= 0 && !is_stop(tok_eos) && g_nstop < 64) g_stop[g_nstop++] = tok_eos;
|
||||
int nsp = 0;
|
||||
if (T) for (int id = 0; id < T->n_ids && g_nstop < 64; id++)
|
||||
if (T->id_special[id] && !is_stop(id)) { g_stop[g_nstop++] = id; nsp++; }
|
||||
/* #401: in serve mode keep ONLY <|endoftext|>. Role markers <|user|>/<|observation|>
|
||||
* (config stops + tokenizer special set) are boundaries the Python server owns; as
|
||||
* hard stops they cut generation the moment the model opens a <tool_call> block,
|
||||
* because int4 argmax noise picks a stop-token ID over the correct '<' token. */
|
||||
if (getenv("SERVE") && tok_eos >= 0) {
|
||||
int kept = 0;
|
||||
for (int i = 0; i < g_nstop; i++) if (g_stop[i] == tok_eos) g_stop[kept++] = g_stop[i];
|
||||
if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d non-EOS stop tokens (tool-call safety, #401)\n", g_nstop - kept);
|
||||
g_nstop = kept; nsp = 0;
|
||||
}
|
||||
fprintf(stderr, "[stop] %d stop tokens:", g_nstop);
|
||||
for (int i = 0; i < g_nstop; i++) fprintf(stderr, " %d", g_stop[i]);
|
||||
if (nsp) fprintf(stderr, " (%d from the tokenizer's special set)", nsp);
|
||||
fprintf(stderr, "\n");
|
||||
}
|
||||
static void stops_arm(const Cfg *c, int tok_eos){ stops_arm_tok(c, tok_eos, NULL); }
|
||||
|
||||
/* ---- log-prob of a target token given the logit vector ------------------- */
|
||||
static double logprob_target(const float *lo, int V, int target, int *am){
|
||||
float mx = lo[0]; int best = 0;
|
||||
for (int i = 1; i < V; i++) if (lo[i] > mx) { mx = lo[i]; best = i; }
|
||||
double se = 0;
|
||||
for (int i = 0; i < V; i++) se += exp((double)lo[i] - mx);
|
||||
if (am) *am = (best == target);
|
||||
return (double)(lo[target] - mx) - log(se);
|
||||
}
|
||||
|
||||
/* "glm" in model_type, case-insensitive */
|
||||
static int mt_is_glm(const char *s){
|
||||
if (s) for (; *s; s++)
|
||||
if ((s[0]|32) == 'g' && (s[1]|32) == 'l' && (s[2]|32) == 'm') return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif /* SAMPLE_H */
|
||||
+2
-2
@@ -32,11 +32,11 @@ esac
|
||||
|
||||
# 2) build: nativa (veloce, per QUESTA macchina). Per un binario da distribuire: make portable
|
||||
echo " building (ARCH=${ARCH:-native})…"
|
||||
make -s glm ARCH="${ARCH:-native}"
|
||||
make -s colibri ARCH="${ARCH:-native}"
|
||||
|
||||
# 3) self-test sull'oracolo tiny, se presente
|
||||
if [ -d glm_tiny ] && [ -f ref_glm.json ]; then
|
||||
r=$(SNAP=./glm_tiny TF=1 ./glm 64 16 16 2>/dev/null | grep -oE "[0-9]+/[0-9]+ positions" || true)
|
||||
r=$(SNAP=./glm_tiny TF=1 ./colibri 64 16 16 2>/dev/null | grep -oE "[0-9]+/[0-9]+ positions" || true)
|
||||
echo " engine self-test: ${r:-?} (expected 32/32)"
|
||||
fi
|
||||
|
||||
|
||||
@@ -40,6 +40,9 @@ typedef struct {
|
||||
int dfds[512]; /* gemelli O_DIRECT (aperti pigramente): -2 = non ancora provato */
|
||||
char *paths[512];
|
||||
int nfd;
|
||||
int mfds[512]; /* MIRROR: fds of the second model copy (dual-SSD), -1 = absent */
|
||||
int mdfds[512]; /* O_DIRECT twins of the second copy, -1 = absent */
|
||||
int nmirror; /* files accepted into the mirror (0 = mirror inactive) */
|
||||
int *hidx; /* hash map nome->indice (open addressing): con ~120k tensori
|
||||
* (GLM: 256 expert x 78 layer x 3 x 2) la scansione lineare
|
||||
* costava decine di secondi/token (misurato sul primo run reale) */
|
||||
@@ -99,10 +102,80 @@ static int st_open_fd(shards *S, const char *path) {
|
||||
|
||||
/* fd gemello O_DIRECT dello stesso file (bypassa la page cache: il buffered read su
|
||||
* ext4-in-VHDX si strozza a ~0.8 GB/s, O_DIRECT arriva a 2.3+; misurato). -1 se non disponibile. */
|
||||
static int st_direct_fd(shards *S, int fd) {
|
||||
for (int i = 0; i < S->nfd; i++) if (S->fds[i] == fd) return S->dfds[i];
|
||||
static int st_fidx(shards *S, int fd) {
|
||||
for (int i = 0; i < S->nfd; i++) if (S->fds[i] == fd) return i;
|
||||
return -1;
|
||||
}
|
||||
static int st_direct_fd(shards *S, int fd) {
|
||||
int i = st_fidx(S, fd); return i < 0 ? -1 : S->dfds[i];
|
||||
}
|
||||
|
||||
/* ---- MIRROR (dual-SSD): second read-only copy of the model on another drive ----
|
||||
* st_fd_rep/st_direct_fd_rep: fd of replica `rep` (0 = primary, 1 = mirror) for
|
||||
* the SAME file identified by its primary fd. -1 if that replica is absent. */
|
||||
static int st_fd_rep(shards *S, int fd, int rep) {
|
||||
if (!rep) return fd;
|
||||
if (!S->nmirror) return -1;
|
||||
int i = st_fidx(S, fd); return i < 0 ? -1 : S->mfds[i];
|
||||
}
|
||||
static int st_direct_fd_rep(shards *S, int fd, int rep) {
|
||||
if (!rep) return st_direct_fd(S, fd);
|
||||
if (!S->nmirror) return -1;
|
||||
int i = st_fidx(S, fd); return i < 0 ? -1 : S->mdfds[i];
|
||||
}
|
||||
|
||||
/* Registers <dir>/<basename> as a read replica of every already-indexed shard.
|
||||
* A file is accepted ONLY if its size and safetensors header are byte-identical
|
||||
* to the primary: the data_offsets then match by construction, so every pread
|
||||
* is valid on either copy. Missing or divergent files simply stay on the
|
||||
* primary (the mirror may be partial, e.g. a smaller SSD holding only the
|
||||
* expert shards). Returns the number of accepted files. The mirror is NEVER
|
||||
* written to: .coli_usage/.coli_kv keep deriving from the primary alone. */
|
||||
static int st_mirror_init(shards *S, const char *dir) {
|
||||
if (S->nmirror) for (int i = 0; i < S->nfd; i++) { /* re-init: drop the old replica */
|
||||
if (S->mfds[i] >= 0) close(S->mfds[i]);
|
||||
if (S->mdfds[i] >= 0) close(S->mdfds[i]);
|
||||
}
|
||||
for (int i = 0; i < ST_MAX_SHARDS; i++) { S->mfds[i] = -1; S->mdfds[i] = -1; }
|
||||
S->nmirror = 0;
|
||||
for (int i = 0; i < S->nfd; i++) {
|
||||
const char *base = strrchr(S->paths[i], '/');
|
||||
#ifdef _WIN32
|
||||
const char *b2 = strrchr(S->paths[i], '\\');
|
||||
if (b2 && (!base || b2 > base)) base = b2;
|
||||
#endif
|
||||
base = base ? base + 1 : S->paths[i];
|
||||
char mp[2048]; snprintf(mp, sizeof(mp), "%s/%s", dir, base);
|
||||
int mfd = open(mp, COMPAT_O_RDONLY);
|
||||
if (mfd < 0) continue; /* partial mirror: this shard stays on the primary */
|
||||
int64_t sza = lseek(S->fds[i], 0, SEEK_END), szb = lseek(mfd, 0, SEEK_END);
|
||||
if (sza != szb) {
|
||||
fprintf(stderr, "[MIRROR] %s: size differs from the primary copy — file skipped\n", mp);
|
||||
close(mfd); continue;
|
||||
}
|
||||
uint64_t ha = 0, hb = 0; int ok = 1; /* identical header => identical data_offsets */
|
||||
if (pread(S->fds[i], &ha, 8, 0) != 8 || pread(mfd, &hb, 8, 0) != 8 ||
|
||||
ha != hb || ha == 0 || ha > (uint64_t)256 << 20 || (int64_t)(8 + ha) > sza) ok = 0;
|
||||
if (ok) {
|
||||
char *ba = malloc(ha), *bb = malloc(ha);
|
||||
if (!ba || !bb || pread(S->fds[i], ba, ha, 8) != (ssize_t)ha ||
|
||||
pread(mfd, bb, ha, 8) != (ssize_t)ha || memcmp(ba, bb, ha)) ok = 0;
|
||||
free(ba); free(bb);
|
||||
}
|
||||
if (!ok) {
|
||||
fprintf(stderr, "[MIRROR] %s: header differs from the primary copy — file skipped\n", mp);
|
||||
close(mfd); continue;
|
||||
}
|
||||
S->mfds[i] = mfd;
|
||||
#ifdef O_DIRECT
|
||||
S->mdfds[i] = open(mp, COMPAT_O_RDONLY | O_DIRECT);
|
||||
#elif defined(__APPLE__)
|
||||
S->mdfds[i] = compat_open_direct(mp);
|
||||
#endif
|
||||
S->nmirror++;
|
||||
}
|
||||
return S->nmirror;
|
||||
}
|
||||
|
||||
/* indicizza tutti i model-*.safetensors in snap_dir */
|
||||
/* pread completo: chunk-loop (una singola pread si ferma a ~2^31 byte su Linux
|
||||
@@ -137,21 +210,66 @@ static void st_pread_full(int fd, void *buf, int64_t n, int64_t off, const char
|
||||
}
|
||||
}
|
||||
|
||||
static void st_init(shards *S, const char *snap_dir) {
|
||||
/* Scan one directory for *.safetensors shards, appending to files[] (dedup by
|
||||
* basename, so a list of directories acts as a SEARCH PATH: the same shard
|
||||
* present on two drives is taken from the first-listed one only). *added
|
||||
* returns how many shards this dir contributed. */
|
||||
static void st_scan_dir(const char *dir, char files[][1024], int *nf, int *added) {
|
||||
DIR *d = opendir(dir); struct dirent *e;
|
||||
if (!d) { perror(dir); exit(1); }
|
||||
int base_n = *nf;
|
||||
while ((e = readdir(d))) {
|
||||
const char *dot = strrchr(e->d_name, '.');
|
||||
if (dot && !strcmp(dot, ".safetensors")) { /* model.safetensors o model-0000N-of-... */
|
||||
int dup = 0;
|
||||
for (int i = 0; i < *nf; i++) {
|
||||
const char *b = strrchr(files[i], '/');
|
||||
#ifdef _WIN32
|
||||
const char *b2 = strrchr(files[i], '\\'); if (b2 && (!b || b2 > b)) b = b2;
|
||||
#endif
|
||||
b = b ? b + 1 : files[i];
|
||||
if (!strcmp(b, e->d_name)) { dup = 1; break; } /* already taken from a higher-priority drive */
|
||||
}
|
||||
if (dup) continue;
|
||||
if (*nf >= ST_MAX_SHARDS) { fprintf(stderr, "too many shards (>%d): raise ST_MAX_SHARDS\n", ST_MAX_SHARDS); exit(1); }
|
||||
snprintf(files[(*nf)++], 1024, "%s/%s", dir, e->d_name);
|
||||
}
|
||||
}
|
||||
closedir(d);
|
||||
if (added) *added = *nf - base_n;
|
||||
}
|
||||
|
||||
/* Index shards from snap_dir, optionally SPLIT across extra drives listed in
|
||||
* extra_dirs (';' or ',' separated). Each shard lives on exactly ONE drive
|
||||
* (no duplication — unlike the dual-SSD mirror); a demand pread hits whichever
|
||||
* drive holds that shard, so concurrent expert loads parallelise across drives
|
||||
* and combined capacity is used. Scales to N drives. Metadata (config /
|
||||
* tokenizer / .coli_usage / .coli_kv) is read from snap_dir only. */
|
||||
static void st_init_multi(shards *S, const char *snap_dir, const char *extra_dirs) {
|
||||
memset(S, 0, sizeof(*S));
|
||||
S->cap = 4096; S->t = calloc(S->cap, sizeof(st_tensor));
|
||||
/* raccoglie ordinatamente i nomi dei file shard */
|
||||
static char files[ST_MAX_SHARDS][1024]; int nf = 0;
|
||||
DIR *d = opendir(snap_dir); struct dirent *e;
|
||||
if (!d) { perror(snap_dir); exit(1); }
|
||||
while ((e = readdir(d))) {
|
||||
const char *dot = strrchr(e->d_name, '.');
|
||||
if (dot && !strcmp(dot, ".safetensors")) { /* model.safetensors o model-0000N-of-... */
|
||||
if (nf >= ST_MAX_SHARDS) { fprintf(stderr, "too many shards (>%d): raise ST_MAX_SHARDS\n", ST_MAX_SHARDS); exit(1); }
|
||||
snprintf(files[nf++], 1024, "%s/%s", snap_dir, e->d_name);
|
||||
int c0 = 0; st_scan_dir(snap_dir, files, &nf, &c0);
|
||||
int ndir = 1;
|
||||
if (extra_dirs && *extra_dirs) {
|
||||
char buf[4096]; snprintf(buf, sizeof(buf), "%s", extra_dirs);
|
||||
char *p = buf;
|
||||
while (p && *p) {
|
||||
char *sep = p; while (*sep && *sep != ';' && *sep != ',') sep++;
|
||||
int last = (*sep == 0); *sep = 0;
|
||||
while (*p == ' ') p++;
|
||||
size_t plen = strlen(p); while (plen > 0 && p[plen-1] == ' ') p[--plen] = 0;
|
||||
if (*p) {
|
||||
int cN = 0; st_scan_dir(p, files, &nf, &cN);
|
||||
fprintf(stderr, "[SPLIT] +%s -> %d shard(s)\n", p, cN);
|
||||
ndir++;
|
||||
}
|
||||
p = last ? NULL : sep + 1;
|
||||
}
|
||||
fprintf(stderr, "[SPLIT] model across %d dir(s): %d shard(s) total (primary %s -> %d shard(s)), no duplication\n",
|
||||
ndir, nf, snap_dir, c0);
|
||||
}
|
||||
closedir(d);
|
||||
for (int a = 0; a < nf; a++) for (int b = a+1; b < nf; b++)
|
||||
if (strcmp(files[a], files[b]) > 0) { char tmp[1024]; strcpy(tmp, files[a]); strcpy(files[a], files[b]); strcpy(files[b], tmp); }
|
||||
|
||||
@@ -200,7 +318,19 @@ static void st_init(shards *S, const char *snap_dir) {
|
||||
if (a0 < 0 || b0 < a0 || data_start + b0 > fsz) {
|
||||
fprintf(stderr, "%s: tensor '%s' data_offsets [%lld,%lld] out of file bounds (%lld)\n",
|
||||
files[fi], name, (long long)a0, (long long)b0, (long long)fsz); exit(1); }
|
||||
int64_t numel = 1; for (int k = 0; k < shp->len; k++) numel *= (int64_t)shp->kids[k]->num;
|
||||
/* SEC: lo shape viene da un file non fidato (mirror). Senza il guard
|
||||
* di overflow, uno shape tipo [65535,65535,65535,...] fa avvolgere
|
||||
* numel a un valore piccolo/negativo che poi passerebbe il cross-check
|
||||
* numel*esz==nbytes in st_read_f32, riaprendo l'OOB. */
|
||||
int64_t numel = 1; int bad_shape = 0;
|
||||
for (int k = 0; k < shp->len; k++) {
|
||||
int64_t d = (int64_t)shp->kids[k]->num;
|
||||
if (d < 0 || (d != 0 && numel > INT64_MAX / d)) { bad_shape = 1; break; }
|
||||
numel *= d;
|
||||
}
|
||||
if (bad_shape) {
|
||||
fprintf(stderr, "%s: tensor '%s' shape overflows int64 — refusing (hostile or corrupt file)\n",
|
||||
files[fi], name); exit(1); }
|
||||
if (S->n == S->cap) { S->cap *= 2; S->t = realloc(S->t, S->cap*sizeof(st_tensor)); }
|
||||
st_tensor *t = &S->t[S->n++];
|
||||
t->name = strdup(name); t->fd = fd; t->off = data_start + a0;
|
||||
@@ -220,6 +350,9 @@ static void st_init(shards *S, const char *snap_dir) {
|
||||
}
|
||||
}
|
||||
|
||||
/* backward-compatible single-directory entry point */
|
||||
static void st_init(shards *S, const char *snap_dir) { st_init_multi(S, snap_dir, NULL); }
|
||||
|
||||
static st_tensor *st_find(shards *S, const char *name) {
|
||||
if (S->hidx) {
|
||||
uint64_t h = st_hash(name) & (S->hcap - 1);
|
||||
@@ -244,11 +377,30 @@ static void st_prefetch(shards *S, const char *name) {
|
||||
if (t) posix_fadvise(t->fd, t->off, t->nbytes, POSIX_FADV_WILLNEED);
|
||||
}
|
||||
|
||||
/* like st_prefetch, but on replica `rep`'s drive: the WILLNEED must warm the
|
||||
* page cache of the SAME fd the later demand pread will hit. */
|
||||
static void st_prefetch_rep(shards *S, const char *name, int rep) {
|
||||
st_tensor *t = st_find(S, name);
|
||||
if (!t) return;
|
||||
int fd = st_fd_rep(S, t->fd, rep);
|
||||
if (fd < 0) fd = t->fd;
|
||||
posix_fadvise(fd, t->off, t->nbytes, POSIX_FADV_WILLNEED);
|
||||
}
|
||||
|
||||
/* legge un tensore in un buffer float32 fornito dal chiamante (numel float).
|
||||
* drop=1 -> consiglia al kernel di scartare le pagine (per gli expert in streaming). */
|
||||
static int64_t st_read_f32(shards *S, const char *name, float *out, int drop) {
|
||||
st_tensor *t = st_find(S, name);
|
||||
if (!t) { fprintf(stderr, "missing tensor: %s\n", name); exit(1); }
|
||||
/* SEC: numel viene dallo shape, nbytes dagli offset — due campi indipendenti
|
||||
* del file. Se non concordano, la memcpy F32 (nbytes) o i loop BF16/F16
|
||||
* (numel elementi da un raw di soli nbytes) sforano il buffer del chiamante,
|
||||
* che e' dimensionato sul config, non sul file. Il chiamante che alloca su
|
||||
* st_numel resta coerente; questo blocca l'ingresso ostile a monte. */
|
||||
int esz = (t->dtype == 2) ? 4 : 2;
|
||||
if (t->numel < 0 || t->numel > t->nbytes / esz || t->numel * (int64_t)esz != t->nbytes) {
|
||||
fprintf(stderr, "%s: tensor '%s' shape/bytes mismatch (numel %lld, %lld bytes, dtype %d) — refusing (hostile or corrupt file)\n",
|
||||
name, name, (long long)t->numel, (long long)t->nbytes, t->dtype); exit(1); }
|
||||
void *raw = malloc(t->nbytes);
|
||||
if (!raw) { fprintf(stderr, "malloc %lld bytes for tensor %s failed\n", (long long)t->nbytes, name); exit(1); }
|
||||
st_pread_full(t->fd, raw, t->nbytes, t->off, "pread data");
|
||||
|
||||
+189
@@ -0,0 +1,189 @@
|
||||
/* telemetry.h — dashboard protocol lines, stats/usage persistence, hardware probe.
|
||||
* Include after Model/Cfg/QT/ESlot/shards and st.h are defined; requires
|
||||
* qt_bytes(), now_s(), rss_gb(), edisk_s(), and the g_cuda_* globals (ifdef). */
|
||||
#ifndef TELEMETRY_H
|
||||
#define TELEMETRY_H
|
||||
|
||||
static int64_t tbytes(int O,int I,int bits){
|
||||
if(bits>=16) return (int64_t)O*I*4;
|
||||
if(bits>=5) return (int64_t)O*I + (int64_t)O*4;
|
||||
return (int64_t)O*((I+1)/2) + (int64_t)O*4;
|
||||
}
|
||||
|
||||
static int64_t expert_bytes_probe(Model *m, int ebits){
|
||||
Cfg *c=&m->c; int64_t eb=0; char nm[256];
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.gate_proj.weight",c->first_dense);
|
||||
if(st_nbytes(&m->S,nm)>0){
|
||||
const char *suf[3]={"gate_proj","up_proj","down_proj"};
|
||||
for(int k=0;k<3;k++){
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.%s.weight",c->first_dense,suf[k]);
|
||||
eb+=st_nbytes(&m->S,nm);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.%s.weight.qs",c->first_dense,suf[k]);
|
||||
int64_t q=st_nbytes(&m->S,nm); if(q>0) eb+=q;
|
||||
}
|
||||
}
|
||||
if(eb<=0) eb = tbytes(c->moe_inter,c->hidden,ebits)*2 + tbytes(c->hidden,c->moe_inter,ebits);
|
||||
return eb;
|
||||
}
|
||||
|
||||
/* BRAIN MAP: per-turn expert hit bitmap for the dashboard. */
|
||||
static uint8_t **g_ehit;
|
||||
static void ehit_mark(Model *m, int layer, int eid){
|
||||
if(!g_ehit){ Cfg *c=&m->c;
|
||||
g_ehit=calloc(c->n_layers+1,sizeof(uint8_t*));
|
||||
for(int i=0;i<=c->n_layers;i++) g_ehit[i]=calloc(c->n_experts,1);
|
||||
}
|
||||
g_ehit[layer][eid]=1;
|
||||
}
|
||||
|
||||
/* CPU model + cores + RAM (GB); empty/zero where unavailable. */
|
||||
static void hw_probe(char *cpu, size_t cn, int *cores, double *ram_total, double *ram_avail){
|
||||
cpu[0]=0;
|
||||
#ifdef _WIN32
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
{ unsigned int r[12]={0}; unsigned int *w=r;
|
||||
for(unsigned int f=0x80000002u; f<=0x80000004u; f++,w+=4)
|
||||
__get_cpuid(f,&w[0],&w[1],&w[2],&w[3]);
|
||||
char *b=(char*)r; b[47]=0; while(*b==' ')b++;
|
||||
snprintf(cpu,cn,"%s",b); }
|
||||
#endif
|
||||
#else
|
||||
FILE *ci=fopen("/proc/cpuinfo","r");
|
||||
if(ci){ char ln[256];
|
||||
while(fgets(ln,sizeof(ln),ci)) if(!strncmp(ln,"model name",10)){
|
||||
char *p=strchr(ln,':'); if(p){ p++; while(*p==' ')p++;
|
||||
int n=(int)strlen(p); if(n>0&&p[n-1]=='\n')p[--n]=0;
|
||||
snprintf(cpu,cn,"%s",p); } break; }
|
||||
fclose(ci); }
|
||||
#endif
|
||||
*cores=0;
|
||||
#ifdef _WIN32
|
||||
{ SYSTEM_INFO si; GetSystemInfo(&si); *cores=(int)si.dwNumberOfProcessors; }
|
||||
#elif defined(_SC_NPROCESSORS_ONLN)
|
||||
*cores=(int)sysconf(_SC_NPROCESSORS_ONLN);
|
||||
#endif
|
||||
*ram_total=*ram_avail=0;
|
||||
#ifdef _WIN32
|
||||
compat_meminfo(ram_total,ram_avail);
|
||||
#else
|
||||
FILE *mi=fopen("/proc/meminfo","r");
|
||||
if(mi){ char ln[256]; double mt=0,ma=0;
|
||||
while(fgets(ln,sizeof(ln),mi)){
|
||||
if(sscanf(ln,"MemTotal: %lf",&mt)==1) *ram_total=mt/1e6;
|
||||
if(sscanf(ln,"MemAvailable: %lf",&ma)==1) *ram_avail=ma/1e6;
|
||||
} fclose(mi); }
|
||||
#endif
|
||||
}
|
||||
|
||||
static void hwinfo_emit(Model *m){
|
||||
Cfg *c=&m->c; (void)c;
|
||||
char cpu[256]; int cores; double ram_total,ram_avail;
|
||||
hw_probe(cpu,sizeof(cpu),&cores,&ram_total,&ram_avail);
|
||||
int ngpu=0; double vram_total=0;
|
||||
char gpu_name[128]="";
|
||||
#ifdef COLI_CUDA
|
||||
ngpu=g_cuda_ndev; vram_total=m->gpu_expert_bytes/1e9;
|
||||
for(int i=0;i<g_cuda_ndev;i++){
|
||||
size_t fr=0,to=0; coli_cuda_mem_info(g_cuda_devices[i],&fr,&to);
|
||||
if(!i) vram_total=(double)to*g_cuda_ndev/1e9;
|
||||
}
|
||||
if(g_cuda_ndev>0)
|
||||
snprintf(gpu_name,sizeof(gpu_name),"CUDA device x%d",g_cuda_ndev);
|
||||
#endif
|
||||
printf("HWINFO %d %.1f %.1f %d %.1f %s|%s\n",
|
||||
cores,ram_total,ram_avail,ngpu,vram_total,cpu,gpu_name);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
static void tiers_emit(Model *m){
|
||||
Cfg *c=&m->c; int nsp=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) nsp++;
|
||||
int total=(nsp+(m->has_mtp?1:0))*c->n_experts;
|
||||
int pinned=0,lru=0;
|
||||
for(int i=0;i<=c->n_layers;i++){ pinned+=m->npin?m->npin[i]:0; lru+=m->ecn?m->ecn[i]:0; }
|
||||
int vram=0; double vram_gb=0;
|
||||
#ifdef COLI_CUDA
|
||||
vram=m->gpu_expert_count; vram_gb=m->gpu_expert_bytes/1e9;
|
||||
#endif
|
||||
int ram=pinned-vram+lru; if(ram<0) ram=0;
|
||||
int disk=total-vram-ram; if(disk<0) disk=0;
|
||||
double eb=(double)expert_bytes_probe(m,m->ebits);
|
||||
printf("TIERS %d %d %d %.2f %.2f\n",vram,ram,disk,vram_gb,ram*eb/1e9);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
static void emap_emit(Model *m){
|
||||
Cfg *c=&m->c;
|
||||
int rows=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) rows++;
|
||||
int has_mtp = m->has_mtp && m->eusage[c->n_layers];
|
||||
if(has_mtp) rows++;
|
||||
int cols=c->n_experts;
|
||||
char *hex=malloc((size_t)rows*cols*2+1); int w=0;
|
||||
for(int i=0;i<=c->n_layers;i++){
|
||||
int is_row = (i<c->n_layers && m->L[i].sparse) || (i==c->n_layers && has_mtp);
|
||||
if(!is_row) continue;
|
||||
for(int e=0;e<cols;e++){
|
||||
int tier=0;
|
||||
ESlot *P=m->pin[i];
|
||||
for(int z=0;z<m->npin[i];z++) if(P[z].eid==e){
|
||||
#ifdef COLI_CUDA
|
||||
tier = P[z].g.cuda?2:1;
|
||||
#else
|
||||
tier = 1;
|
||||
#endif
|
||||
break; }
|
||||
if(!tier && m->ecache && m->ecache[i])
|
||||
for(int z=0;z<m->ecn[i];z++) if(m->ecache[i][z].eid==e){ tier=1; break; }
|
||||
uint32_t u = m->eusage[i]?m->eusage[i][e]:0;
|
||||
int heat=0; while(u){ heat++; u>>=1; } if(heat>63) heat=63;
|
||||
int b=(tier<<6)|heat;
|
||||
hex[w++]="0123456789abcdef"[b>>4]; hex[w++]="0123456789abcdef"[b&15];
|
||||
}
|
||||
}
|
||||
hex[w]=0;
|
||||
printf("EMAP %d %d %s\n",rows,cols,hex); fflush(stdout); free(hex);
|
||||
}
|
||||
|
||||
static void hits_emit(Model *m){
|
||||
Cfg *c=&m->c; if(!g_ehit) return;
|
||||
int rows=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) rows++;
|
||||
int has_mtp = m->has_mtp && m->eusage[c->n_layers];
|
||||
if(has_mtp) rows++;
|
||||
int cols=c->n_experts, nb=(rows*cols+7)/8;
|
||||
uint8_t *bm=calloc(nb,1); int bit=0;
|
||||
for(int i=0;i<=c->n_layers;i++){
|
||||
int is_row = (i<c->n_layers && m->L[i].sparse) || (i==c->n_layers && has_mtp);
|
||||
if(!is_row) continue;
|
||||
for(int e=0;e<cols;e++,bit++)
|
||||
if(g_ehit[i][e]){ bm[bit>>3]|=1<<(bit&7); g_ehit[i][e]=0; }
|
||||
}
|
||||
char *hex=malloc((size_t)nb*2+1); int w=0;
|
||||
for(int b=0;b<nb;b++){ hex[w++]="0123456789abcdef"[bm[b]>>4]; hex[w++]="0123456789abcdef"[bm[b]&15]; }
|
||||
hex[w]=0;
|
||||
printf("HITS %d %d %s\n",rows,cols,hex); fflush(stdout); free(hex); free(bm);
|
||||
}
|
||||
|
||||
static void stats_dump_q(Model *m, const char *path, int quiet){
|
||||
char tmp[2100]; snprintf(tmp,sizeof(tmp),"%s.tmp",path);
|
||||
FILE *f=fopen(tmp,"w"); if(!f){ if(!quiet) perror(tmp); return; }
|
||||
Cfg *c=&m->c; int64_t tot=0, nz=0;
|
||||
for(int i=0;i<=c->n_layers;i++){ if(!m->eusage[i]) continue;
|
||||
for(int e=0;e<c->n_experts;e++) if(m->eusage[i][e]){ fprintf(f,"%d %d %u\n",i,e,m->eusage[i][e]); tot+=m->eusage[i][e]; nz++; } }
|
||||
fclose(f); rename(tmp,path);
|
||||
if(!quiet) fprintf(stderr,"[STATS] %lld selections across %lld distinct experts -> %s\n",(long long)tot,(long long)nz,path);
|
||||
}
|
||||
static void stats_dump(Model *m, const char *path){ stats_dump_q(m,path,0); }
|
||||
|
||||
static char g_usage_path[2100]="";
|
||||
static int64_t usage_load(Model *m, const char *path){
|
||||
FILE *f=fopen(path,"r"); if(!f) return 0;
|
||||
Cfg *c=&m->c; int l,e; uint32_t cnt; int64_t tot=0;
|
||||
while(fscanf(f,"%d %d %u",&l,&e,&cnt)==3)
|
||||
if(l>=0&&l<=c->n_layers&&e>=0&&e<c->n_experts&&m->eusage[l]){ m->eusage[l][e]+=cnt; tot+=cnt; }
|
||||
fclose(f); return tot;
|
||||
}
|
||||
static void usage_save(Model *m){ if(g_usage_path[0]) stats_dump_q(m,g_usage_path,1); }
|
||||
|
||||
#endif /* TELEMETRY_H */
|
||||
@@ -0,0 +1,98 @@
|
||||
# Efficiency suite — regression tests + optimization dossier
|
||||
|
||||
Two layers:
|
||||
|
||||
1. **`test_inefficiency.py`** — tiny-model *asserted* regression tests. Fast
|
||||
(~0.15s/run), gate CI, catch breakage. Run as part of `make test`.
|
||||
2. **`test_efficiency_report.py`** — an *opt-in optimization dossier* for a real
|
||||
model. Runs every instrumentation flag, prints a 9-section report answering
|
||||
*what is doing what, when, with what, is it inefficient, how to improve*.
|
||||
Never fails CI (it's a report, not a gate).
|
||||
|
||||
## The dossier (what you run when optimizing)
|
||||
|
||||
```bash
|
||||
# CPU-only (safe, fast to validate):
|
||||
COLI_EFFICIENCY_MODEL=../glm52_i4_g64 make efficiency-report
|
||||
|
||||
# CUDA (dense + expert tiers — needs a CUDA build, see below):
|
||||
COLI_EFFICIENCY_MODEL=../glm52_i4_g64 COLI_EFFICIENCY_CUDA=1 make efficiency-report
|
||||
```
|
||||
|
||||
It turns ON every observability flag the engine supports — `PROF=1`,
|
||||
`COLI_CUDA_PROFILE=1`, `CACHE_ROUTE=1` (auto-unlocks `route_agree`/`route_kl`),
|
||||
`DISK_SPLIT=1`, `LOOKA=1` — so nothing the engine can tell you is left dark.
|
||||
None of these change the computed output; they only add telemetry.
|
||||
|
||||
The 9 sections, and the question each answers:
|
||||
|
||||
| § | section | answers |
|
||||
|---|---|---|
|
||||
| 1 | PROVENANCE | what is running, on what CPU/backend, with what effective config |
|
||||
| 2 | THROUGHPUT | tok/s + forward-latency p50/p90/p99/max (is the tail healthy?) |
|
||||
| 3 | WHERE TIME GOES | the 5 PROFILE phases as % of decode + absolute seconds + verdict |
|
||||
| 3a | ATTENTION BREAKDOWN | attention split into projection/RoPE, score-softmax-value, output |
|
||||
| 4 | EXPERT CACHE | hit %, experts-loaded/token vs baseline topk |
|
||||
| 5 | DISK I/O | GB fetched, MB/token, GB/s, read-service vs felt-wait, phase split |
|
||||
| 5a | DISK-LOAD SPLIT | loads by decode phase (draft/absorb/verify) + MTP-vs-main bytes |
|
||||
| 6 | ROUTING QUALITY | route_agree %, route_kl, cache swaps |
|
||||
| 6a | ROUTING PREDICTABILITY | LOOKAHEAD recall per predictor (which prefetch wins) |
|
||||
| 7 | SPECULATION | tokens/forward, MTP acceptance % |
|
||||
| 8 | GPU TIERS | resident tensors, expert tier (count/GB/calls), H2D/kernel/D2H ms |
|
||||
|
||||
Every line that crosses an advisory threshold is marked `[FLAG]` with the
|
||||
concrete lever to pull (raise RAM_GB, add PIN_GB, try DIRECT=1, lower CTX, …),
|
||||
and all flags repeat in a summary at the end.
|
||||
|
||||
## Tunable thresholds
|
||||
|
||||
The `IS IT INEFFICIENT?` lines are advisory constants at the top of
|
||||
`test_efficiency_report.py`:
|
||||
|
||||
| constant | default | meaning |
|
||||
|---|---|---|
|
||||
| `DISK_WAIT_DOMINANT` | 0.40 | >40% decode waiting on expert reads → I/O-bound |
|
||||
| `LOW_HIT_RATE` | 0.30 | <30% cache hit → thrashing |
|
||||
| `LOW_ROUTE_AGREE` | 0.80 | <80% routing overlap → prefetch guessing wrong |
|
||||
| `HIGH_TAIL_RATIO` | 3.0 | p99 > 3× p50 → decode stalls |
|
||||
| `LOW_MTP_ACCEPT` | 0.20 | <20% MTP acceptance → draft decoder is dead weight |
|
||||
|
||||
The tiny-model asserted floors live in `tools/efficiency.py` (`TINY_TOK_S_FLOOR`,
|
||||
`MAX_DISK_WAIT_SHARE`, `MIN_CPU_CUDA_AGREEMENT`).
|
||||
|
||||
## The regression tests (what gates CI)
|
||||
|
||||
`test_inefficiency.py` runs on the bundled `glm_tiny` model and asserts:
|
||||
|
||||
- telemetry parses (no format drift)
|
||||
- tiny tok/s ≥ floor (throughput regression)
|
||||
- PROFILE phases present and non-negative (accounting sanity)
|
||||
- disk-wait not dominant on a resident model (I/O-path regression)
|
||||
- CPU determinism (two greedy runs agree)
|
||||
- **CUDA** (skip unless CUDA built): init path, dense uploads VRAM, CPU-vs-CUDA
|
||||
argmax agreement ≥ 70% (kernel-correctness guard)
|
||||
|
||||
```bash
|
||||
make efficiency # tiny CPU tests
|
||||
make efficiency-cuda # tiny CUDA tests (needs CUDA build)
|
||||
```
|
||||
|
||||
## CUDA build prerequisite
|
||||
|
||||
The default `make glm.exe` builds **without** CUDA. The CUDA tests and the CUDA
|
||||
dossier need a host built with `-DCOLI_CUDA` plus the runtime DLL:
|
||||
|
||||
```bash
|
||||
make clean && make glm.exe CUDA_DLL=1 && make cuda-dll
|
||||
```
|
||||
|
||||
`make efficiency-cuda` auto-skips with a clear message if the host is CPU-only
|
||||
(it scans the binary for the "CPU-only" marker the engine embeds).
|
||||
|
||||
## Files
|
||||
|
||||
- `tools/efficiency.py` — shared harness: `parse_run()` (captures every
|
||||
telemetry signal), `run_engine()`, thresholds. Reuses `PROFILE_RE`/`SPEED_RE`
|
||||
from `tools/benchmark_cuda_fixture.py`.
|
||||
- `tests/test_inefficiency.py` — tiny-model asserted tests (CPU + CUDA).
|
||||
- `tests/test_efficiency_report.py` — the opt-in optimization dossier.
|
||||
@@ -24,7 +24,7 @@
|
||||
* Run: make tests/bench_dsa_select && ./tests/bench_dsa_select (not in TEST_BINS)
|
||||
*/
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
/* Microbenchmark: old (single-accumulator) vs new (independent-accumulator) AVX-VNNI
|
||||
* int8/int4 dot kernels (quant.h). NOT a unit test -- test_idot.c proves correctness.
|
||||
* This measures the headline claim: breaking the serial vpdpbusd->acc chain lifts
|
||||
* per-core kernel throughput, the same win the NEON path already took ("26->63 GB/s, 2.4x").
|
||||
*
|
||||
* It re-implements the OLD single-acc AVX-VNNI kernels inline and calls the REAL (new)
|
||||
* ones via the include-colibri.c pattern -- one process, identical frozen inputs, warm
|
||||
* caches. Reports median ns/call + GB/s-of-weights + new/old ratio. Measures the kernel's
|
||||
* compute ceiling (warm caches), NOT end-to-end tok/s.
|
||||
*
|
||||
* Run: make tests/bench_idot ARCH=native && ./tests/bench_idot (not in TEST_BINS -- not a gate)
|
||||
*/
|
||||
#define main coli_glm_main_unused
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
|
||||
static uint32_t rs=0x2545F491u;
|
||||
static uint32_t xr(void){ rs^=rs<<13; rs^=rs>>17; rs^=rs<<5; return rs; }
|
||||
|
||||
/* ---- OLD kernels: verbatim copies of the pre-change __AVXVNNI__ branches (single acc) ---- */
|
||||
#if defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
static int32_t dot_i8i8_old(const int8_t *w, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
__m128i acc=_mm_setzero_si128();
|
||||
for(;i+16<=I;i+=16){
|
||||
__m128i wv=_mm_loadu_si128((const __m128i*)(w+i));
|
||||
__m128i xv=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i xs=_mm_sign_epi8(xv,wv);
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
for(;i<I;i++) sum+=(int32_t)w[i]*x[i];
|
||||
return sum;
|
||||
}
|
||||
static int32_t dot_i4i8_old(const uint8_t *w4, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8);
|
||||
__m128i acc=_mm_setzero_si128();
|
||||
for(;i+32<=I;i+=32){
|
||||
__m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m128i w0=_mm_sub_epi8(n0,b8), w1=_mm_sub_epi8(n1,b8);
|
||||
__m128i x0=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
for(;i<I;i++){ uint8_t b=w4[i>>1]; int v=(i&1)?((int)(b>>4)-8):((int)(b&0xF)-8); sum+=v*x[i]; }
|
||||
return sum;
|
||||
}
|
||||
#else
|
||||
#error "bench_idot requires an AVX-VNNI build: make tests/bench_idot ARCH=native on an AVX-VNNI CPU"
|
||||
#endif
|
||||
|
||||
#define I_DIM 6144
|
||||
#define N_REPEAT 20000
|
||||
static int cmp_d(const void*a,const void*b){ double x=*(const double*)a,y=*(const double*)b; return x<y?-1:x>y?1:0; }
|
||||
|
||||
int main(void){
|
||||
static int8_t w8[I_DIM], x8[I_DIM]; static uint8_t w4[I_DIM/2];
|
||||
for(int i=0;i<I_DIM;i++){ w8[i]=(int8_t)(xr()&0xFF); x8[i]=(int8_t)((int)(xr()%255)-127); }
|
||||
for(int i=0;i<I_DIM/2;i++) w4[i]=(uint8_t)(xr()&0xFF);
|
||||
|
||||
/* correctness sanity: new must equal old (bit-exact) */
|
||||
if(dot_i8i8(w8,x8,I_DIM)!=dot_i8i8_old(w8,x8,I_DIM)){ fprintf(stderr,"MISMATCH i8i8\n"); return 1; }
|
||||
if(dot_i4i8(w4,x8,I_DIM)!=dot_i4i8_old(w4,x8,I_DIM)){ fprintf(stderr,"MISMATCH i4i8\n"); return 1; }
|
||||
|
||||
static double t[N_REPEAT]; volatile int32_t sink=0;
|
||||
const char *names[4]={"i8i8 old","i8i8 new","i4i8 old","i4i8 new"};
|
||||
double gbs[4];
|
||||
for(int k=0;k<4;k++){
|
||||
for(int wi=0;wi<200;wi++) sink+= (k==0)?dot_i8i8_old(w8,x8,I_DIM):(k==1)?dot_i8i8(w8,x8,I_DIM)
|
||||
:(k==2)?dot_i4i8_old(w4,x8,I_DIM):dot_i4i8(w4,x8,I_DIM); /* warmup */
|
||||
for(int r=0;r<N_REPEAT;r++){
|
||||
double t0=now_s();
|
||||
sink+= (k==0)?dot_i8i8_old(w8,x8,I_DIM):(k==1)?dot_i8i8(w8,x8,I_DIM)
|
||||
:(k==2)?dot_i4i8_old(w4,x8,I_DIM):dot_i4i8(w4,x8,I_DIM);
|
||||
t[r]=(now_s()-t0)*1e9;
|
||||
}
|
||||
qsort(t,N_REPEAT,sizeof(double),cmp_d);
|
||||
double med=t[N_REPEAT/2];
|
||||
double bytes=(k<2)?I_DIM:(double)I_DIM/2; /* weight bytes touched */
|
||||
gbs[k]=bytes/med;
|
||||
printf("%-9s %8.1f ns/call %6.2f GB/s\n", names[k], med, gbs[k]);
|
||||
}
|
||||
printf("ratio i8i8 new/old: %.2fx | ratio i4i8 new/old: %.2fx\n", gbs[1]/gbs[0], gbs[3]/gbs[2]);
|
||||
(void)sink; return 0;
|
||||
}
|
||||
@@ -30,9 +30,9 @@ int main(){
|
||||
for(int i=0;i<D;i++)ds[i]=0.006f+(i%7)*0.0002f;
|
||||
for(size_t i=0;i<x.size();i++)x[i]=std::sin((float)(i+1)*0.013f)*2.f;
|
||||
ColiCudaTensor *g=nullptr,*u=nullptr,*d=nullptr;
|
||||
if(!coli_cuda_tensor_upload(&g,hidden.data(),hs.data(),2,D,I,device)||
|
||||
!coli_cuda_tensor_upload(&u,hidden.data(),hs.data(),2,D,I,device)||
|
||||
!coli_cuda_tensor_upload(&d,down.data(),ds.data(),2,I,D,device))return 2;
|
||||
if(!coli_cuda_tensor_upload(&g,hidden.data(),hs.data(),2,D,I,device,0)||
|
||||
!coli_cuda_tensor_upload(&u,hidden.data(),hs.data(),2,D,I,device,0)||
|
||||
!coli_cuda_tensor_upload(&d,down.data(),ds.data(),2,I,D,device,0))return 2;
|
||||
for(int rows: {1,2,4,8}){
|
||||
double scalar=run(g,u,d,x.data(),a.data(),rows,3,0);
|
||||
double packed=run(g,u,d,x.data(),b.data(),rows,3,1);
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
* Run: make tests/bench_topp && ./tests/bench_topp (not in TEST_BINS -- not a gate)
|
||||
*/
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -50,10 +50,56 @@ int main(int argc, char **argv) {
|
||||
if (coli_cuda_tensor_upload(&t8, q8, s8, 1, 5, 2, d0)) return 1;
|
||||
if (ndev > 1 && coli_cuda_tensor_upload(&t8, q8, s8, 1, 4, 2, d1)) return 1;
|
||||
if (!coli_cuda_matmul(&t8, got, x, q8, s8, 1, 2, 4, 2, d0) || !close_enough(got, want8, 4)) return 1;
|
||||
/* Cached tensor must stay callable without live host pointers
|
||||
* (CUDA_RELEASE_HOST slots null theirs after upload) — including
|
||||
* SUSTAINED reuse, not just the first call. */
|
||||
for (int rep = 0; rep < 64; rep++)
|
||||
if (!coli_cuda_matmul(&t8, got, x, nullptr, nullptr, 1, 2, 4, 2, d0) ||
|
||||
!close_enough(got, want8, 4)) return 1;
|
||||
/* A tensor uploaded from a TEMPORARY host buffer must survive the buffer
|
||||
* being scribbled and freed (the release-host lifecycle). */
|
||||
{
|
||||
int8_t *tmpw = static_cast<int8_t *>(std::malloc(8));
|
||||
float *tmps = static_cast<float *>(std::malloc(2 * sizeof(float)));
|
||||
if (!tmpw || !tmps) return 2;
|
||||
for (int i = 0; i < 8; i++) tmpw[i] = q8[i];
|
||||
tmps[0] = s8[0]; tmps[1] = s8[1];
|
||||
ColiCudaTensor *tt = nullptr;
|
||||
if (!coli_cuda_tensor_upload(&tt, tmpw, tmps, 1, 4, 2, d0)) return 1;
|
||||
for (int i = 0; i < 8; i++) tmpw[i] = 99;
|
||||
std::free(tmpw); std::free(tmps);
|
||||
if (!coli_cuda_matmul(&tt, got, x, nullptr, nullptr, 1, 2, 4, 2, d0) ||
|
||||
!close_enough(got, want8, 4)) return 1;
|
||||
coli_cuda_tensor_free(tt);
|
||||
}
|
||||
/* Upload failures must be graceful and must not corrupt accounting —
|
||||
* and must not poison LATER healthy launches (sticky-error regression). */
|
||||
{
|
||||
size_t c0 = 0, b0 = 0, c1 = 0, b1 = 0;
|
||||
coli_cuda_stats(-1, &c0, &b0);
|
||||
ColiCudaTensor *bad = nullptr;
|
||||
if (coli_cuda_tensor_upload(&bad, q8, s8, 1, 4, 2, 9999)) return 1;
|
||||
if (coli_cuda_tensor_upload(&bad, q8, s8, 7, 4, 2, d0)) return 1;
|
||||
if (coli_cuda_tensor_upload(&bad, q8, nullptr, 1, 4, 2, d0)) return 1;
|
||||
if (coli_cuda_tensor_upload(&bad, nullptr, s8, 1, 4, 2, d0)) return 1;
|
||||
if (coli_cuda_tensor_upload(&bad, q8, s8, 1, 1 << 20, 1 << 24, d0)) return 1; /* ~16 TB */
|
||||
if (bad) return 1;
|
||||
coli_cuda_stats(-1, &c1, &b1);
|
||||
if (c0 != c1 || b0 != b1) return 1;
|
||||
/* healthy launch immediately after the failed allocation */
|
||||
if (!coli_cuda_matmul(&t8, got, x, nullptr, nullptr, 1, 2, 4, 2, d0) ||
|
||||
!close_enough(got, want8, 4)) return 1;
|
||||
}
|
||||
/* Fault injection hook: on/off, restores cleanly. */
|
||||
if (setenv("COLI_GPU_FAIL_AFTER", "0", 1)) return 2;
|
||||
if (coli_cuda_matmul(&t8, got, x, nullptr, nullptr, 1, 2, 4, 2, d0)) return 1;
|
||||
if (unsetenv("COLI_GPU_FAIL_AFTER")) return 2;
|
||||
if (!coli_cuda_matmul(&t8, got, x, nullptr, nullptr, 1, 2, 4, 2, d0) ||
|
||||
!close_enough(got, want8, 4)) return 1;
|
||||
const int8_t q8b[8]={-1,-2,-3,-4, 1,-2,3,-4};
|
||||
const float s8b[2]={1.f,.5f},want8b[4]={10.f,15.f,-3.f,-2.5f};
|
||||
if(!coli_cuda_tensor_update(t8,q8b,s8b)||
|
||||
!coli_cuda_matmul(&t8,got,x,q8b,s8b,1,2,4,2,d0)||
|
||||
!coli_cuda_matmul(&t8,got,x,q8b,s8b,1,2,4,2,d0,0)||
|
||||
!close_enough(got,want8b,4))return 1;
|
||||
|
||||
/* Rows [-8,-1,0,7] and [1,2,3,4], packed low nibble first. */
|
||||
@@ -61,26 +107,26 @@ int main(int argc, char **argv) {
|
||||
const float s4[2] = {1.0f, 0.25f};
|
||||
const float want4[2] = {-34.0f, -2.5f};
|
||||
ColiCudaTensor *t4 = nullptr;
|
||||
if (!coli_cuda_matmul(&t4, got, x, q4, s4, 2, 1, 4, 2, d1) || !close_enough(got, want4, 2)) return 1;
|
||||
if (!coli_cuda_matmul(&t4, got, x, q4, s4, 2, 1, 4, 2, d1, 0) || !close_enough(got, want4, 2)) return 1;
|
||||
|
||||
const uint8_t q2[2] = {0xe4, 0x1b};
|
||||
const float s2[2] = {0.5f, 2.0f};
|
||||
const float want2[2] = {-2.0f, 12.0f};
|
||||
ColiCudaTensor *t2 = nullptr;
|
||||
if (!coli_cuda_matmul(&t2, got, x, q2, s2, 3, 1, 4, 2, d1) || !close_enough(got, want2, 2)) return 1;
|
||||
if (!coli_cuda_matmul(&t2, got, x, q2, s2, 3, 1, 4, 2, d1, 0) || !close_enough(got, want2, 2)) return 1;
|
||||
|
||||
const float wf[8] = {1, 0, -1, 2, 0.5f, 0.5f, 0.5f, 0.5f};
|
||||
const float wantf[2] = {-10.0f, -1.0f};
|
||||
ColiCudaTensor *tf = nullptr;
|
||||
if (!coli_cuda_matmul(&tf, got, x, wf, nullptr, 0, 1, 4, 2, d0) || !close_enough(got, wantf, 2)) return 1;
|
||||
if (!coli_cuda_matmul(&tf, got, x, wf, nullptr, 0, 1, 4, 2, d0, 0) || !close_enough(got, wantf, 2)) return 1;
|
||||
|
||||
const float eg[8] = {1,0,0,0, 0,1,0,0};
|
||||
const float eu[8] = {1,0,0,0, 0,1,0,0};
|
||||
const float ed[8] = {1,0, 0,1, 1,1, 1,-1};
|
||||
ColiCudaTensor *tg=nullptr,*tu=nullptr,*td=nullptr;
|
||||
if (!coli_cuda_tensor_upload(&tg,eg,nullptr,0,4,2,d0) ||
|
||||
!coli_cuda_tensor_upload(&tu,eu,nullptr,0,4,2,d0) ||
|
||||
!coli_cuda_tensor_upload(&td,ed,nullptr,0,2,4,d0)) return 1;
|
||||
if (!coli_cuda_tensor_upload(&tg,eg,nullptr,0,4,2,d0,0) ||
|
||||
!coli_cuda_tensor_upload(&tu,eu,nullptr,0,4,2,d0,0) ||
|
||||
!coli_cuda_tensor_upload(&td,ed,nullptr,0,2,4,d0,0)) return 1;
|
||||
float expert[8], want_expert[8];
|
||||
for(int s=0;s<2;s++){
|
||||
float a=x[s*4], b=x[s*4+1];
|
||||
@@ -98,7 +144,7 @@ int main(int argc, char **argv) {
|
||||
const float aw[16]={1,0,0,0, 0,1,0,0, 0,0,1,0, 0,0,0,1};
|
||||
const float aq[4]={1,2,.5f,-.5f},al[12]={1,0,0,0, 0,1,0,0, 0,0,1,0};
|
||||
const float ar[6]={1,0, 0,1, 1,1};float actx[2],aref[2];
|
||||
ColiCudaTensor *at=nullptr;if(!coli_cuda_tensor_upload(&at,aw,nullptr,0,4,4,d0))return 1;
|
||||
ColiCudaTensor *at=nullptr;if(!coli_cuda_tensor_upload(&at,aw,nullptr,0,4,4,d0,0))return 1;
|
||||
float score[3];for(int t=0;t<3;t++)score[t]=aq[0]*al[t*4]+aq[1]*al[t*4+1]+aq[2]*ar[t*2]+aq[3]*ar[t*2+1];
|
||||
float mx=score[0],z=0;for(int t=1;t<3;t++)mx=score[t]>mx?score[t]:mx;
|
||||
for(int t=0;t<3;t++){score[t]=std::exp(score[t]-mx);z+=score[t];}for(int t=0;t<3;t++)score[t]/=z;
|
||||
@@ -117,9 +163,9 @@ int main(int argc, char **argv) {
|
||||
for(int i=0;i<32;i++)ws4[i]=0.01f+(i%5)*0.002f;
|
||||
for(int i=0;i<64;i++)gx4[i]=std::sin((float)(i+1)*0.17f)*2.f;
|
||||
ColiCudaTensor *g4=nullptr,*u4=nullptr,*d4=nullptr;
|
||||
if(!coli_cuda_tensor_upload(&g4,w4,ws4,2,32,32,d0)||
|
||||
!coli_cuda_tensor_upload(&u4,w4,ws4,2,32,32,d0)||
|
||||
!coli_cuda_tensor_upload(&d4,w4,ws4,2,32,32,d0))return 1;
|
||||
if(!coli_cuda_tensor_upload(&g4,w4,ws4,2,32,32,d0,0)||
|
||||
!coli_cuda_tensor_upload(&u4,w4,ws4,2,32,32,d0,0)||
|
||||
!coli_cuda_tensor_upload(&d4,w4,ws4,2,32,32,d0,0))return 1;
|
||||
ColiCudaTensor *gg4[2]={g4,g4},*ug4[2]={u4,u4},*dg4[2]={d4,d4};
|
||||
if(!coli_cuda_expert_group(gg4,ug4,dg4,group_rows,2,scalar4,gx4))return 1;
|
||||
setenv("COLI_CUDA_TC_INT4","1",1);
|
||||
|
||||
@@ -177,6 +177,68 @@ static int run_attn(int S, int pos_base, const char* name){
|
||||
return pass?0:1;
|
||||
}
|
||||
|
||||
// serial r_top8 vs parallel r_top8_par on the ENGINE build's own compiled shaders — the
|
||||
// exact-match contract (same indices, same order, same weights bitwise, same keff)
|
||||
// enforced with memcmp, per adversarial input family. `mode` selects the input
|
||||
// construction; see the inventory at the call sites in main(). E is a parameter (not
|
||||
// hardcoded 256) so the same helper drives both the original E=256 fuzz and the
|
||||
// expert-count-generality cases (E=24 <32-lane-width, E=168 REAP-pruned, E=200
|
||||
// lane-straddling boundary, E=257 out-of-contract auto-serial-fallback proof).
|
||||
static int run_rtop8(int mode, int S, int E, float topp, int normk, float rscale, const char *name) {
|
||||
const int K=8, Ksel=8;
|
||||
std::vector<float> sig((size_t)S*E), bias(E);
|
||||
srand(4242+mode*17+S+E);
|
||||
for (int e=0;e<E;e++) bias[e]=((rand()%2001)-1000)/1000.f;
|
||||
for (int s=0;s<S;s++) for (int e=0;e<E;e++) {
|
||||
float *v=&sig[(size_t)s*E+e];
|
||||
switch (mode) {
|
||||
case 0: *v=(float)(rand()%10000)/10000.f; break; // generic sigmoid-like
|
||||
case 1: *v=0.5f; break; // ALL EQUAL: pure tie-break test
|
||||
case 2: *v=(float)((e/2)%8)/8.f; break; // massed duplicates (paired+cyclic ties)
|
||||
case 3: *v=(e%2)?1e-40f:2e-40f; break; // denormal logits (flush behavior must match)
|
||||
case 4: *v=(float)(rand()%3)/2.f; break; // 3-level ties across the whole row
|
||||
// boundary-forcing: elevate the LAST 4 valid experts (E-4..E-1) to near-max choice
|
||||
// so they are guaranteed in the top-8. For an E whose per-lane block size doesn't
|
||||
// divide E evenly, E-1's lane straddles the E boundary (real indices below E,
|
||||
// sentinel -1e30f at/above E in the SAME ch[] block) -- e.g. E=200: per=ceil(200/
|
||||
// 32)=7, lane 28 owns indices 196..202, of which 196-199 are real and 200-202 are
|
||||
// sentinel. Forcing selection onto 196-199 exercises exactly that lane's per-index
|
||||
// e<E boundary check, rather than hoping random data happens to land there.
|
||||
case 5: *v=(e>=E-4)?1.0f:(float)(rand()%10000)/10000.f; break;
|
||||
default: *v=(float)(rand()%10000)/10000.f; break;
|
||||
}
|
||||
}
|
||||
if (mode==1) for (int e=0;e<E;e++) bias[e]=0.25f; // choice fully tied too
|
||||
if (mode==3) for (int e=0;e<E;e++) bias[e]=(e%3)?3e-40f:-3e-40f; // denormal bias as well
|
||||
if (mode==5) for (int e=E-4;e<E;e++) bias[e]=1.0f; // combined choice = 2.0, max possible
|
||||
std::vector<int> is((size_t)S*K), ip((size_t)S*K); std::vector<float> ws((size_t)S*K), wp((size_t)S*K);
|
||||
std::vector<int> ks(S), kp(S);
|
||||
if (!coli_metal_rtop8(0,sig.data(),bias.data(),S,E,K,Ksel,topp,normk,rscale,is.data(),ws.data(),ks.data()) ||
|
||||
!coli_metal_rtop8(1,sig.data(),bias.data(),S,E,K,Ksel,topp,normk,rscale,ip.data(),wp.data(),kp.data())) {
|
||||
printf(" %-34s FAIL (rtop8 runner returned 0)\n", name); return 1; }
|
||||
int ok = memcmp(is.data(),ip.data(),(size_t)S*K*4)==0 &&
|
||||
memcmp(ws.data(),wp.data(),(size_t)S*K*4)==0 && // bitwise: same ops, same order
|
||||
memcmp(ks.data(),kp.data(),(size_t)S*4)==0;
|
||||
if (mode==5 && ok) {
|
||||
// Don't just trust the input design -- confirm the straddling lane's valid segment
|
||||
// (E-4..E-1) was actually selected, in EVERY row, so this case can't silently
|
||||
// degrade into an unrelated pass if the input construction above ever changes.
|
||||
for (int s=0;s<S;s++) { int seen=0;
|
||||
for (int k=0;k<K;k++) if (ip[(size_t)s*K+k]>=E-4 && ip[(size_t)s*K+k]<E) seen++;
|
||||
if (seen<4) { printf(" %-34s *** boundary segment not exercised (row %d saw %d/4) -- test setup bug\n", name, s, seen); return 1; }
|
||||
}
|
||||
}
|
||||
if (!ok) {
|
||||
printf(" %-34s *** MISMATCH\n", name);
|
||||
for (int s=0;s<S;s++){ printf(" row %d keff %d/%d:",s,ks[s],kp[s]);
|
||||
for(int k=0;k<K;k++) printf(" [%d]%d/%d %.6g/%.6g",k,is[s*K+k],ip[s*K+k],ws[s*K+k],wp[s*K+k]);
|
||||
printf("\n"); }
|
||||
return 1;
|
||||
}
|
||||
printf(" %-34s ok (serial==parallel bitwise, S=%d E=%d)\n", name, S, E);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
if (!coli_metal_init()) { printf("Metal unavailable (skipping)\n"); return 0; }
|
||||
printf("Metal backend kernel tests:\n");
|
||||
@@ -215,6 +277,32 @@ int main(void) {
|
||||
fail |= run_attn(1, 37, "attn S=1 pos=37");
|
||||
fail |= run_attn(4, 12, "attn S=4 pos=12 (MTP)");
|
||||
fail |= run_attn(3, 0, "attn S=3 pos=0");
|
||||
printf("Metal top-8 select serial-vs-parallel tests (exact-match contract, E=256):\n");
|
||||
fail |= run_rtop8(0, 1, 256, 0.0f, 1, 1.0f, "top8 generic S=1");
|
||||
fail |= run_rtop8(0, 4, 256, 0.0f, 1, 1.0f, "top8 generic S=4");
|
||||
fail |= run_rtop8(1, 1, 256, 0.0f, 1, 1.0f, "top8 ALL-EQUAL ties");
|
||||
fail |= run_rtop8(2, 4, 256, 0.0f, 1, 1.0f, "top8 massed dup ties S=4");
|
||||
fail |= run_rtop8(4, 2, 256, 0.0f, 0, 2.5f, "top8 3-level ties rscale");
|
||||
fail |= run_rtop8(3, 1, 256, 0.0f, 1, 1.0f, "top8 denormal logits");
|
||||
fail |= run_rtop8(0, 1, 256, 0.01f, 1, 1.0f, "top8 topp=0.01 (Ke=1 edge)");
|
||||
fail |= run_rtop8(2, 1, 256, 0.6f, 1, 1.0f, "top8 topp=0.6 tied weights");
|
||||
fail |= run_rtop8(0, 4, 256, 0.999f,1, 1.75f, "top8 topp=0.999 S=4");
|
||||
fail |= run_rtop8(1, 2, 256, 0.5f, 0, 1.0f, "top8 topp on ALL-EQUAL");
|
||||
printf("Metal top-8 select expert-count-generality tests (E!=256, REAP/#428 motivated):\n");
|
||||
fail |= run_rtop8(0, 1, 168, 0.0f, 1, 1.0f, "top8 E=168 (REAP) generic S=1");
|
||||
fail |= run_rtop8(2, 4, 168, 0.0f, 1, 1.0f, "top8 E=168 (REAP) massed dup ties S=4");
|
||||
fail |= run_rtop8(0, 1, 24, 0.0f, 1, 1.0f, "top8 E=24 (<32 lane width) generic");
|
||||
fail |= run_rtop8(1, 1, 24, 0.0f, 1, 1.0f, "top8 E=24 (<32 lane width) ALL-EQUAL ties");
|
||||
// E=200: per-lane block size ceil(200/32)=7, and 200 is NOT a multiple of 7, so lane 28
|
||||
// (indices 196..202) straddles the boundary -- 196-199 real, 200-202 sentinel -1e30f in
|
||||
// the SAME ch[] block. E=24 and E=168 above both happen to divide evenly by their own
|
||||
// per (24/1, 168/6), so no case before this one exercised a lane whose ch[] mixes real
|
||||
// and sentinel indices. mode 5 deterministically forces indices 196-199 into the top-8
|
||||
// (see run_rtop8) and asserts they were actually selected, rather than hoping random
|
||||
// data lands there -- proving by TEST what the per-index `e<E` check was proven by
|
||||
// reading (both kernels agree bitwise on a selection that requires that check to fire).
|
||||
fail |= run_rtop8(5, 4, 200, 0.0f, 1, 1.0f, "top8 E=200 (lane straddles E boundary)");
|
||||
fail |= run_rtop8(0, 1, 257, 0.0f, 1, 1.0f, "top8 E=257 (>256, auto-serial-fallback)");
|
||||
printf(fail? "metal backend tests: FAILED\n" : "metal backend tests: ok\n");
|
||||
coli_metal_shutdown();
|
||||
return fail;
|
||||
|
||||
@@ -44,6 +44,12 @@ static void test_submit_header(void)
|
||||
assert(coli_submit_parse("SUBMIT 1 0 16777216 3 1 1", &sub));
|
||||
assert(!coli_submit_parse("SUBMIT 1 0 16777217 3 1 1", &sub));
|
||||
assert(!coli_submit_parse("SUBMIT 1 0 2 3 1 1 trailing", &sub));
|
||||
/* optional 7th field: per-request grammar length (0 when absent) */
|
||||
assert(coli_submit_parse("SUBMIT 42 3 17 64 0.7 0.95", &sub) && sub.gbytes == 0);
|
||||
assert(coli_submit_parse("SUBMIT 42 3 17 64 0.7 0.95 512", &sub) && sub.gbytes == 512);
|
||||
assert(coli_submit_parse("SUBMIT 42 3 17 64 0.7 0.95 1048576", &sub));
|
||||
assert(!coli_submit_parse("SUBMIT 42 3 17 64 0.7 0.95 1048577", &sub));
|
||||
assert(!coli_submit_parse("SUBMIT 42 3 17 64 0.7 0.95 512 extra", &sub));
|
||||
}
|
||||
|
||||
int main(void)
|
||||
|
||||
@@ -31,7 +31,7 @@
|
||||
* In-memory only (no scratch files), so it builds clean on the Windows MinGW CI
|
||||
* job without the unmerged compat shim. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -0,0 +1,315 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Exhaustive optimization dossier for a colibri engine run.
|
||||
|
||||
This is NOT a pass/fail test. It runs the engine with every instrumentation flag
|
||||
on (PROF, COLI_CUDA_PROFILE, CACHE_ROUTE, DISK_SPLIT, LOOKA) and prints a section-
|
||||
by-section report answering, for each subsystem:
|
||||
|
||||
WHAT is doing it — which phase/kernel/tier
|
||||
WHEN it is doing it — how much of decode wall-time it owns
|
||||
WITH WHAT — the config/weights/tier it used
|
||||
IS IT INEFFICIENT? — a verdict, with the threshold
|
||||
HOW TO IMPROVE — the concrete knob, named
|
||||
|
||||
Activation (opt-in only — NOT in `make test`):
|
||||
COLI_EFFICIENCY_MODEL=<model_dir> python tests/test_efficiency_report.py
|
||||
|
||||
Optional env:
|
||||
COLI_EFFICIENCY_CUDA=1 also exercise the CUDA dense/expert tiers
|
||||
COLI_EFFICIENCY_NGEN=N decode tokens (default 24)
|
||||
COLI_EFFICIENCY_RAM_GB=N RAM budget (default 28)
|
||||
COLI_EFFICIENCY_VRAM_GB=N CUDA expert-tier budget GB (default 4)
|
||||
COLI_EFFICIENCY_PROMPT=... prompt (default: a code-gen prompt)
|
||||
|
||||
Exit code is always 0 (it's a dossier, not a gate). Lines marked FLAG point at
|
||||
the most likely lever to move tok/s for the observed bottleneck.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from tools.efficiency import run_engine, disk_wait_share # noqa: E402
|
||||
|
||||
|
||||
# --- advisory thresholds (the "IS IT INEFFICIENT?" lines) ---
|
||||
DISK_WAIT_DOMINANT = 0.40 # >40% decode waiting on expert reads -> I/O-bound
|
||||
LOW_HIT_RATE = 0.30 # <30% cache hit -> thrashing (cap too small)
|
||||
LOW_ROUTE_AGREE = 0.80 # <80% routing overlap -> prefetch is guessing wrong
|
||||
HIGH_TAIL_RATIO = 3.0 # p99 > 3x p50 -> decode stalls (I/O hiccups / KV grow)
|
||||
LOW_MTP_ACCEPT = 0.20 # <20% MTP acceptance -> draft decoder is dead weight
|
||||
VRAM_WASTE_CALLS = 0 # experts pinned in VRAM but 0 calls served
|
||||
|
||||
|
||||
def _flag(ok): return "OK " if ok else "FLAG"
|
||||
|
||||
|
||||
def _bar(frac, width=24):
|
||||
"""A simple ASCII bar for share visualization."""
|
||||
n = max(0, min(width, round(frac * width)))
|
||||
return "#" * n + "." * (width - n)
|
||||
|
||||
|
||||
def _line(label, value, flag=None, note=""):
|
||||
tag = f" [{flag}]" if flag else ""
|
||||
print(f" {label:<22} {value}{tag} {note}" if note else f" {label:<22} {value}{tag}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
model = os.environ.get("COLI_EFFICIENCY_MODEL")
|
||||
if not model:
|
||||
print(__doc__)
|
||||
print("\nNot activated: set COLI_EFFICIENCY_MODEL=<model_dir> to run.")
|
||||
return 0
|
||||
model = str(Path(model).resolve())
|
||||
if not Path(model).is_dir():
|
||||
print(f"ERROR: {model} is not a directory", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
ngen = int(os.environ.get("COLI_EFFICIENCY_NGEN", "24"))
|
||||
ram_gb = os.environ.get("COLI_EFFICIENCY_RAM_GB", "28")
|
||||
vram_gb = os.environ.get("COLI_EFFICIENCY_VRAM_GB", "4")
|
||||
prompt = os.environ.get(
|
||||
"COLI_EFFICIENCY_PROMPT",
|
||||
"Write a Python function that computes the factorial of a number. "
|
||||
"Include error handling and a docstring.")
|
||||
use_cuda = os.environ.get("COLI_EFFICIENCY_CUDA") == "1"
|
||||
|
||||
# Turn ON every instrumentation flag so the dossier has maximum detail.
|
||||
# These are all observability toggles (PROF/COLI_CUDA_PROFILE/CACHE_ROUTE/
|
||||
# DISK_SPLIT/LOOKA); none change the computed output.
|
||||
overlay = dict(
|
||||
NGEN=str(ngen), TEMP="0", RAM_GB=ram_gb, PROMPT=prompt,
|
||||
PROF="1", CACHE_ROUTE="1", DISK_SPLIT="1", LOOKA="1", ROUTE_AGREE="1",
|
||||
)
|
||||
if use_cuda:
|
||||
overlay.update(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1",
|
||||
COLI_CUDA_PROFILE="1", CUDA_EXPERT_GB=vram_gb)
|
||||
|
||||
print("=" * 78)
|
||||
print(f"OPTIMIZATION DOSSIER — {Path(model).name}")
|
||||
print(f" mode : {'CUDA (dense+expert tiers)' if use_cuda else 'CPU-only'} "
|
||||
f"ngen : {ngen} ram : {ram_gb} GB" +
|
||||
(f" vram : {vram_gb} GB" if use_cuda else ""))
|
||||
print("=" * 78)
|
||||
|
||||
t0 = time.time()
|
||||
t, proc = run_engine(overlay, snap=model, timeout=3600.0)
|
||||
wall = time.time() - t0
|
||||
flags = [] # collected FLAG lines for the summary
|
||||
|
||||
print(f"\n[0] RUN")
|
||||
_line("wall clock", f"{wall:.0f}s")
|
||||
_line("exit code", proc.returncode,
|
||||
None if proc.returncode == 0 else "FLAG",
|
||||
"" if proc.returncode == 0 else "non-zero exit")
|
||||
if proc.returncode != 0:
|
||||
print(" stderr tail:")
|
||||
for ln in proc.stderr.strip().splitlines()[-8:]:
|
||||
print(f" {ln}")
|
||||
return 0
|
||||
|
||||
# ---------------------------------------------------------------- [1] WHO ----
|
||||
print(f"\n[1] PROVENANCE — what is running, on what, with what config")
|
||||
if t.get("machine"):
|
||||
m = t["machine"]
|
||||
_line("CPU", m["cpu"])
|
||||
_line("cores / omp", f"{m['cores']} cores")
|
||||
_line("backend", m["backend"])
|
||||
if t.get("load"):
|
||||
ld = t["load"]
|
||||
_line("model load time", f"{ld['load_s']:.2f}s")
|
||||
_line("resident dense", f"{ld['resident_dense_mb']:.1f} MB")
|
||||
_line("layers / experts", f"{ld['layers']} layers, {ld['experts']} experts")
|
||||
_line("MTP", f"{ld['mtp_status']} (draft={ld['draft']})")
|
||||
if t.get("config_str"):
|
||||
_line("resolved config", t["config_str"])
|
||||
print(" (this is the EFFECTIVE config after auto-budgeting — not your env verbatim)")
|
||||
|
||||
# ---------------------------------------------------------------- [2] SPEED --
|
||||
print(f"\n[2] THROUGHPUT — is it fast, is the tail healthy")
|
||||
if t.get("tok_s") is not None:
|
||||
_line("tok/s", f"{t['tok_s']:.3f}")
|
||||
else:
|
||||
flags.append("throughput line missing — engine output format may have changed")
|
||||
if t.get("latency"):
|
||||
la = t["latency"]
|
||||
_line("decode forwards", f"{int(la['forwards'])}")
|
||||
_line("p50 / p90", f"{la['p50_ms']:.1f} / {la['p90_ms']:.1f} ms")
|
||||
_line("p99 / max", f"{la['p99_ms']:.1f} / {la['max_ms']:.1f} ms")
|
||||
tail_ok = la["p99_ms"] <= HIGH_TAIL_RATIO * la["p50_ms"]
|
||||
_line("tail ratio (p99/p50)", f"{la['p99_ms']/max(la['p50_ms'],1e-9):.2f}x",
|
||||
_flag(tail_ok),
|
||||
"high tail = decode stalls (I/O hiccups, KV growth, re-pin)")
|
||||
if not tail_ok:
|
||||
flags.append(f"tail latency p99={la['p99_ms']:.1f}ms >> p50={la['p50_ms']:.1f}ms "
|
||||
"(look for REPIN swaps or disk stalls)")
|
||||
|
||||
# ---------------------------------------------------------------- [3] TIME ---
|
||||
print(f"\n[3] WHERE TIME GOES — what is doing it, when (share of decode)")
|
||||
ts = t.get("time_shares")
|
||||
prof = t.get("profile")
|
||||
if ts:
|
||||
order = [("io", "expert-disk I/O", DISK_WAIT_DOMINANT, "the cache is too small / disk is slow"),
|
||||
("matmul", "expert matmul", 0.40, "compute-bound; more cores or a GPU expert tier"),
|
||||
("attention", "attention", 0.35, "context length is the cost; lower CTX"),
|
||||
("head", "lm_head", 0.10, "vocab projection; unusual to dominate"),
|
||||
("other", "other", 0.30, "scheduling / KV bookkeeping overhead")]
|
||||
for key, name, thresh, lever in order:
|
||||
f = ts[key]
|
||||
ok = f < thresh
|
||||
_line(name, f"{f:5.0%} {_bar(f)}", _flag(ok),
|
||||
"" if ok else f"->{lever}")
|
||||
if not ok:
|
||||
flags.append(f"{name} dominates ({f:.0%}) -> {lever}")
|
||||
if t.get("verdict"):
|
||||
print(f" engine verdict : {t['verdict']}")
|
||||
elif prof:
|
||||
print(" (no [PROF] time shares — set PROF=1 for phase percentages)")
|
||||
if prof:
|
||||
print(" absolute seconds :")
|
||||
for k in ("disk", "expert_matmul", "attention", "lm_head", "other"):
|
||||
_line(k, f"{prof[k]:.3f}s")
|
||||
|
||||
# attention sub-breakdown: how is attention being read
|
||||
ab = t.get("attn_breakdown")
|
||||
if ab:
|
||||
print(f"\n[3a] ATTENTION BREAKDOWN — how the attention phase is spent")
|
||||
atot = sum(ab.values()) or 1.0
|
||||
for k, label in (("proj_rope", "projection + RoPE"),
|
||||
("score_sm_value", "score-softmax-value"),
|
||||
("out_proj", "output projection")):
|
||||
_line(label, f"{ab[k]:.3f}s ({ab[k]/atot:.0%} of attn)")
|
||||
|
||||
# ---------------------------------------------------------------- [4] CACHE --
|
||||
print(f"\n[4] EXPERT CACHE — is the cache efficient")
|
||||
hit = t.get("hit_pct")
|
||||
if hit is not None:
|
||||
ok = hit >= LOW_HIT_RATE * 100
|
||||
_line("hit rate", f"{hit:.1f}%", _flag(ok),
|
||||
"" if ok else "<30% = thrashing; raise RAM_GB or cap")
|
||||
if not ok:
|
||||
flags.append(f"cache hit {hit:.1f}% is low -> raise RAM_GB (or cap), add PIN_GB")
|
||||
el = t.get("experts_loaded")
|
||||
if el:
|
||||
per_tok = el["per_tok"]
|
||||
_line("experts loaded/token", f"{per_tok:.1f}")
|
||||
_line(" per-layer", f"{el['per_layer']:.2f} across {el['n_sparse_layers']} sparse layers")
|
||||
_line(" baseline", f"topk={el['baseline_topk']} active experts/token")
|
||||
base_topk = el["baseline_topk"]
|
||||
if base_topk > 0 and per_tok > 2 * base_topk:
|
||||
flags.append(f"loading {per_tok:.0f} experts/token vs topk={base_topk} "
|
||||
"-> redundant I/O; cache is re-fetching evicted experts")
|
||||
|
||||
# ---------------------------------------------------------------- [5] DISK ---
|
||||
print(f"\n[5] DISK I/O — is I/O the bottleneck, and where")
|
||||
eio = t.get("expert_io")
|
||||
if eio:
|
||||
_line("total fetched", f"{eio['gb_fetched']:.3f} GB")
|
||||
_line("per token", f"{eio['mb_per_tok']:.1f} MB/token")
|
||||
_line("disk throughput", f"{eio['gb_per_s']:.2f} GB/s over the run")
|
||||
_line("read service", f"{eio['read_service_s']:.2f}s (on I/O threads)")
|
||||
_line("felt wait", f"{eio['felt_wait_s']:.2f}s (stall compute felt)")
|
||||
if eio["felt_wait_s"] > eio["read_service_s"] * 0.5 and eio["read_service_s"] > 0:
|
||||
flags.append("felt wait is a large fraction of read service -> PIPE=1 may not be "
|
||||
"overlapping fully, or DIRECT=1 on NVMe")
|
||||
ds = t.get("disk_split")
|
||||
if ds:
|
||||
print(f"\n[5a] DISK-LOAD SPLIT — which decode phase reads the bytes")
|
||||
_line("draft phase", f"{ds['draft']} loads")
|
||||
_line("absorb phase", f"{ds['absorb']} loads")
|
||||
_line("verify/main", f"{ds['verify_main']} loads")
|
||||
_line("MTP-layer bytes", f"{ds['mtp_loads']} loads, {ds['mtp_gb']:.2f} GB")
|
||||
_line("main-layer bytes", f"{ds['main_loads']} loads, {ds['main_gb']:.2f} GB")
|
||||
if ds.get("mtp_bytes_pct") is not None:
|
||||
_line("MTP share of bytes", f"{ds['mtp_bytes_pct']:.1f}%")
|
||||
share = disk_wait_share(t)
|
||||
if share is not None:
|
||||
ok = share < DISK_WAIT_DOMINANT
|
||||
_line("disk-wait share", f"{share:.0%}", _flag(ok),
|
||||
"" if ok else "I/O-bound (see levers in [3])")
|
||||
|
||||
# ---------------------------------------------------------------- [6] ROUTE --
|
||||
print(f"\n[6] ROUTING QUALITY — is the router / prefetch accurate")
|
||||
ra = t.get("route_agree")
|
||||
if ra:
|
||||
ok = ra["agree_pct"] >= LOW_ROUTE_AGREE * 100
|
||||
_line("route_agree", f"{ra['agree_pct']:.1f}% overlap with true top-K",
|
||||
_flag(ok),
|
||||
"" if ok else "prefetch is guessing wrong; CACHE_ROUTE params may need tuning")
|
||||
_line("route_kl", f"{ra['kl']:.4f} mean KL (lower = closer to true routing)")
|
||||
if not ok:
|
||||
flags.append(f"route_agree {ra['agree_pct']:.1f}% low -> tune ROUTE_J/M/P, "
|
||||
"or prefetch is hurting more than helping")
|
||||
sw = t.get("swap")
|
||||
if sw:
|
||||
_line("cache swaps", f"{sw['swaps']}/{sw['slots']} ({sw['pct']:.1f}%)",
|
||||
None, "high swap = churn between turns")
|
||||
la = t.get("lookahead")
|
||||
if la:
|
||||
print(f"\n[6a] ROUTING PREDICTABILITY — recall of true experts in predicted top-8")
|
||||
print(" (which predictor should drive prefetch? highest recall wins)")
|
||||
for row in la:
|
||||
_line(row["predictor"][:34], f"{row['pct']:5.1f}% ({row['hit']}/{row['tot']})")
|
||||
|
||||
# ---------------------------------------------------------------- [7] SPEC ---
|
||||
print(f"\n[7] SPECULATION — is the draft decoder pulling weight")
|
||||
sp = t.get("speculation")
|
||||
if sp:
|
||||
_line("tokens/forward", f"{sp['tok_per_fw']:.2f} (>1.0 means speculation helps)")
|
||||
_line("forwards/tokens", f"{sp['forwards']} forwards for {sp['tokens']} tokens")
|
||||
# Speculation helps only if acceptance is high enough that tok/forward > 1.
|
||||
# tok_per_fw already == 1.0 when nothing verifies, so judge by acceptance.
|
||||
ok = sp["mtp_accept_pct"] >= LOW_MTP_ACCEPT * 100
|
||||
_line("MTP acceptance", f"{sp['mtp_accept_pct']:.0f}%", _flag(ok),
|
||||
"" if ok else "<20% -> drafts rarely verify; DRAFT=0 may be faster")
|
||||
if not ok:
|
||||
flags.append(f"MTP acceptance {sp['mtp_accept_pct']:.0f}% low -> "
|
||||
"drafts cost more I/O than they save; try DRAFT=0")
|
||||
|
||||
# ---------------------------------------------------------------- [8] GPU ----
|
||||
print(f"\n[8] GPU TIERS — is the GPU actually used")
|
||||
cuda = t.get("cuda") or {}
|
||||
if not cuda.get("enabled"):
|
||||
print(" (CUDA not enabled — CPU-only run)")
|
||||
else:
|
||||
if cuda.get("resident_tensors") is not None:
|
||||
_line("resident dense tensors", f"{cuda['resident_tensors']} tensors, "
|
||||
f"{cuda['resident_gb']:.2f} GB")
|
||||
if cuda.get("expert_count") is not None:
|
||||
waste = cuda["calls_served"] <= VRAM_WASTE_CALLS
|
||||
_line("expert tier", f"{cuda['expert_count']} experts pinned "
|
||||
f"({cuda['expert_gb']:.2f} GB)", _flag(not waste))
|
||||
_line(" calls served", f"{cuda['calls_served']} from VRAM",
|
||||
_flag(not waste),
|
||||
"" if not waste else "WASTE: pinned but never routed -> lower CUDA_EXPERT_GB")
|
||||
if waste:
|
||||
flags.append("VRAM expert tier has 0 calls served -> experts pinned but unused; "
|
||||
"PIN stats may not match this workload")
|
||||
if cuda.get("groups"):
|
||||
g = cuda["groups"]
|
||||
_line("expert groups", f"{g['calls']} calls, {g['experts']} experts, "
|
||||
f"{g['rows']} rows ({g['experts_per_call']:.1f} experts/call)")
|
||||
if cuda.get("groups_timing"):
|
||||
gt = cuda["groups_timing"]
|
||||
_line("GPU timing", f"H2D {gt['h2d_ms']:.1f} ms | kernel {gt['kernel_ms']:.1f} ms | "
|
||||
f"D2H {gt['d2h_ms']:.1f} ms")
|
||||
if gt["h2d_ms"] + gt["d2h_ms"] > gt["kernel_ms"]:
|
||||
flags.append("CUDA H2D+D2H > kernel time -> transfer-bound; "
|
||||
"consider larger expert tier to keep weights resident")
|
||||
|
||||
# ---------------------------------------------------------------- summary ----
|
||||
print("\n" + "=" * 78)
|
||||
if flags:
|
||||
print(f" {len(flags)} FLAG(s) — the most likely levers to move tok/s:")
|
||||
for f in flags:
|
||||
print(f" - {f}")
|
||||
else:
|
||||
print(" no flags — every measured subsystem is within advisory thresholds.")
|
||||
print("=" * 78)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -32,9 +32,16 @@ def args(**over):
|
||||
|
||||
|
||||
class EnvDefaultsTest(unittest.TestCase):
|
||||
def env_for_with(self, environ, platform):
|
||||
def env_for_with(self, environ, platform, cuda=False):
|
||||
"""Run env_for on a bare-chat args() under a faked env + platform.
|
||||
|
||||
cuda=False by default so the existing default-I/O tests stay
|
||||
deterministic: the Windows auto-enable branch calls cuda_binary() and
|
||||
(if True) discover_gpus(), both of which reach the real machine — faking
|
||||
False keeps these tests independent of the host's GPU."""
|
||||
with mock.patch.dict(os.environ, environ, clear=True), \
|
||||
mock.patch.object(sys, "platform", platform):
|
||||
mock.patch.object(sys, "platform", platform), \
|
||||
mock.patch.object(coli, "cuda_binary", return_value=cuda):
|
||||
return coli.env_for(args())
|
||||
|
||||
def test_win32_sets_measured_defaults(self):
|
||||
@@ -64,5 +71,80 @@ class EnvDefaultsTest(unittest.TestCase):
|
||||
self.assertNotIn(k, e)
|
||||
|
||||
|
||||
class CudaAutoEnableTest(unittest.TestCase):
|
||||
"""Windows bare `coli chat` (no --gpu/--vram/--auto-tier) used to ALWAYS run
|
||||
CPU-only even on a CUDA build with a GPU present. env_for now auto-enables
|
||||
CUDA on win32 when cuda_binary() is True and a GPU is discoverable; falls
|
||||
back to CPU with a warning if nvidia-smi (discover_gpus) is missing; stays
|
||||
silent on a CPU build; and never touches the Linux path."""
|
||||
|
||||
def _env_for(self, platform, cuda, gpus, plan=None):
|
||||
# Patch discover_gpus / build_plan / environment_for_plan at the
|
||||
# resource_plan module (env_for imports them lazily on each call, so the
|
||||
# patches are live when those imports run). Stubbing the planner keeps
|
||||
# the test independent of a real model dir (args().model == "X").
|
||||
import resource_plan
|
||||
a = args()
|
||||
GPB = 1024 ** 3
|
||||
if plan is None:
|
||||
plan = {"tiers": {"ram": {"budget_bytes": 16 * GPB, "cache_slots_per_layer": 4},
|
||||
"vram": {"budget_bytes": int(8.0 * GPB), "devices": gpus}}}
|
||||
|
||||
def fake_environment_for_plan(p, env, cuda_enabled=True):
|
||||
# Mirror the real contract: size CUDA_EXPERT_GB from the plan's VRAM
|
||||
# budget (this is the value env_for propagates into the engine env).
|
||||
r = dict(env)
|
||||
if cuda_enabled and p["tiers"]["vram"]["devices"] and p["tiers"]["vram"]["budget_bytes"] > 0:
|
||||
r["CUDA_EXPERT_GB"] = f"{p['tiers']['vram']['budget_bytes'] / GPB:.3f}"
|
||||
return r
|
||||
|
||||
with mock.patch.dict(os.environ, {}, clear=True), \
|
||||
mock.patch.object(sys, "platform", platform), \
|
||||
mock.patch.object(coli, "cuda_binary", return_value=cuda), \
|
||||
mock.patch.object(resource_plan, "discover_gpus", return_value=gpus), \
|
||||
mock.patch.object(resource_plan, "build_plan", return_value=plan), \
|
||||
mock.patch.object(resource_plan, "environment_for_plan",
|
||||
side_effect=fake_environment_for_plan):
|
||||
return coli.env_for(a)
|
||||
|
||||
def _fake_gpu(self, index=0, name="NVIDIA GeForce RTX 5070 Ti",
|
||||
total_mib=16384, free_mib=15000):
|
||||
return {"index": index, "name": name,
|
||||
"total_bytes": total_mib * 1024 * 1024,
|
||||
"free_bytes": free_mib * 1024 * 1024}
|
||||
|
||||
def test_win32_auto_enables_cuda_when_gpu_present(self):
|
||||
e = self._env_for("win32", cuda=True, gpus=[self._fake_gpu()])
|
||||
self.assertEqual(e["COLI_CUDA"], "1")
|
||||
self.assertEqual(e["COLI_GPUS"], "0")
|
||||
# VRAM budget is sized from free VRAM by build_plan (real minus reserve),
|
||||
# so it must be present and positive — never a guess or zero.
|
||||
self.assertIn("CUDA_EXPERT_GB", e)
|
||||
self.assertGreater(float(e["CUDA_EXPERT_GB"]), 0.0)
|
||||
# Dense offload is an explicit opt-in (matches --auto-tier): not set here.
|
||||
self.assertNotIn("CUDA_DENSE", e)
|
||||
|
||||
def test_win32_falls_back_to_cpu_when_nvidia_smi_missing(self):
|
||||
# coli_cuda.dll present (cuda=True) but nvidia-smi absent (no GPUs found)
|
||||
# -> warn + CPU-only, never crash, never set COLI_CUDA.
|
||||
e = self._env_for("win32", cuda=True, gpus=[])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("COLI_GPUS", e)
|
||||
self.assertNotIn("CUDA_EXPERT_GB", e)
|
||||
|
||||
def test_win32_cpu_build_stays_silent(self):
|
||||
# No coli_cuda.dll (cuda=False) -> CPU build, nothing GPU-related emitted.
|
||||
e = self._env_for("win32", cuda=False, gpus=[self._fake_gpu()])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("COLI_GPUS", e)
|
||||
|
||||
def test_linux_bare_chat_not_auto_enabled(self):
|
||||
# The auto-enable is scoped to win32: a Linux bare chat with a GPU
|
||||
# present must NOT turn CUDA on (Linux keeps the explicit-flag UX).
|
||||
e = self._env_for("linux", cuda=True, gpus=[self._fake_gpu()])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("CUDA_EXPERT_GB", e)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
/* Grouped-int4 (fmt=4) CUDA kernel oracle (#334).
|
||||
*
|
||||
* Feeds random offset-binary nibble weights + [O, ng] group scales through
|
||||
* grouped_hidden_g4_dual / grouped_down_g4 and checks against a CPU reference
|
||||
* that replicates matmul_i4_grouped's semantics (value = nibble - 8, per-group
|
||||
* partial dot x scale). Covers gs=64, a non-divisible tail group, and a
|
||||
* per-row (gs=0) member riding in the same launch — the fmt=2-compat case.
|
||||
*
|
||||
* The device buffers get the same XOR 0x88 offset->signed conversion the
|
||||
* upload path applies, so the kernels are exercised exactly as deployed.
|
||||
*
|
||||
* Build: nvcc -O2 -std=c++17 -arch=native tests/test_grouped_g4_cuda.cu -o tests/test_grouped_g4
|
||||
*/
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <cmath>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "../backend_cuda.cu"
|
||||
|
||||
static void cpu_gemv_g4(const uint8_t *q,const float *sc,int K,int O,int gs,
|
||||
const float *x,float *y){
|
||||
int rb=(K+1)/2, ng=gs>0?(K+gs-1)/gs:1, egs=gs>0?gs:K;
|
||||
for(int o=0;o<O;o++){
|
||||
const uint8_t *row=q+(size_t)o*rb; const float *scl=sc+(size_t)o*ng;
|
||||
double a=0;
|
||||
for(int g=0; g*egs<K; g++){
|
||||
int base=g*egs, glen=egs; if(base+glen>K) glen=K-base;
|
||||
double p=0;
|
||||
for(int i=base;i<base+glen;i++){
|
||||
uint8_t v=row[i>>1]; int n=(i&1)?(v>>4):(v&15);
|
||||
p+=(double)x[i]*(n-8);
|
||||
}
|
||||
a+=p*scl[g];
|
||||
}
|
||||
y[o]=(float)a;
|
||||
}
|
||||
}
|
||||
|
||||
int main(void){
|
||||
srand(7);
|
||||
const int D=200, I=96, gs=64; /* tail group: 200 % 64 = 8 */
|
||||
const int COUNT=3; /* expert 0,1: fmt4 gs=64; expert 2: per-row (gs=0) */
|
||||
const int rbD=(D+1)/2, rbI=(I+1)/2;
|
||||
const int ngD=(D+gs-1)/gs, ngI=(I+gs-1)/gs;
|
||||
int trials=50, bad=0;
|
||||
for(int t=0;t<trials;t++){
|
||||
GroupDesc host[COUNT]; float *xs; cudaMallocManaged(&xs,(size_t)COUNT*D*4);
|
||||
float *gate,*up,*y; cudaMallocManaged(&gate,(size_t)COUNT*I*4);
|
||||
cudaMallocManaged(&up,(size_t)COUNT*I*4); cudaMallocManaged(&y,(size_t)COUNT*D*4);
|
||||
uint8_t *qg[COUNT],*qu[COUNT],*qd[COUNT]; float *sg[COUNT],*su[COUNT],*sd[COUNT];
|
||||
uint8_t *hg[COUNT],*hu[COUNT],*hd[COUNT]; float *hgs[COUNT],*hus[COUNT],*hds[COUNT];
|
||||
for(int c=0;c<COUNT;c++){
|
||||
int cgs = c==2 ? 0 : gs;
|
||||
int cngD = cgs? ngD:1, cngI = cgs? ngI:1;
|
||||
hg[c]=(uint8_t*)malloc((size_t)I*rbD); hu[c]=(uint8_t*)malloc((size_t)I*rbD);
|
||||
hd[c]=(uint8_t*)malloc((size_t)D*rbI);
|
||||
hgs[c]=(float*)malloc((size_t)I*cngD*4); hus[c]=(float*)malloc((size_t)I*cngD*4);
|
||||
hds[c]=(float*)malloc((size_t)D*cngI*4);
|
||||
for(size_t i=0;i<(size_t)I*rbD;i++){ hg[c][i]=rand()&255; hu[c][i]=rand()&255; }
|
||||
for(size_t i=0;i<(size_t)D*rbI;i++) hd[c][i]=rand()&255;
|
||||
for(size_t i=0;i<(size_t)I*cngD;i++){ hgs[c][i]=.01f+.05f*(rand()/(float)RAND_MAX);
|
||||
hus[c][i]=.01f+.05f*(rand()/(float)RAND_MAX); }
|
||||
for(size_t i=0;i<(size_t)D*cngI;i++) hds[c][i]=.01f+.05f*(rand()/(float)RAND_MAX);
|
||||
cudaMalloc(&qg[c],(size_t)I*rbD); cudaMalloc(&qu[c],(size_t)I*rbD); cudaMalloc(&qd[c],(size_t)D*rbI);
|
||||
cudaMalloc(&sg[c],(size_t)I*cngD*4); cudaMalloc(&su[c],(size_t)I*cngD*4); cudaMalloc(&sd[c],(size_t)D*cngI*4);
|
||||
cudaMemcpy(qg[c],hg[c],(size_t)I*rbD,cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(qu[c],hu[c],(size_t)I*rbD,cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(qd[c],hd[c],(size_t)D*rbI,cudaMemcpyHostToDevice);
|
||||
offset_to_signed_s4<<<64,256>>>(qg[c],(size_t)I*rbD);
|
||||
offset_to_signed_s4<<<64,256>>>(qu[c],(size_t)I*rbD);
|
||||
offset_to_signed_s4<<<64,256>>>(qd[c],(size_t)D*rbI);
|
||||
cudaMemcpy(sg[c],hgs[c],(size_t)I*cngD*4,cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(su[c],hus[c],(size_t)I*cngD*4,cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(sd[c],hds[c],(size_t)D*cngI*4,cudaMemcpyHostToDevice);
|
||||
host[c]={qg[c],qu[c],qd[c],sg[c],su[c],sd[c],4,4,4,1,c,cgs,cgs,cgs};
|
||||
}
|
||||
for(size_t i=0;i<(size_t)COUNT*D;i++) xs[i]=(rand()/(float)RAND_MAX-.5f)*2.f;
|
||||
GroupDesc *ddesc; cudaMalloc(&ddesc,sizeof(host));
|
||||
cudaMemcpy(ddesc,host,sizeof(host),cudaMemcpyHostToDevice);
|
||||
dim3 hgd((unsigned)I,1,(unsigned)COUNT),ogd((unsigned)D,1,(unsigned)COUNT);
|
||||
grouped_hidden_g4_dual<<<hgd,256>>>(gate,up,xs,ddesc,I,D);
|
||||
grouped_down_g4<<<ogd,256>>>(y,gate,ddesc,D,I);
|
||||
if(cudaDeviceSynchronize()!=cudaSuccess){ printf("FAIL cuda\n"); return 1; }
|
||||
for(int c=0;c<COUNT;c++){
|
||||
int cgs=c==2?0:gs;
|
||||
float rg[512],ru[512],ry[512];
|
||||
cpu_gemv_g4(hg[c],hgs[c],D,I,cgs,xs+(size_t)c*D,rg);
|
||||
cpu_gemv_g4(hu[c],hus[c],D,I,cgs,xs+(size_t)c*D,ru);
|
||||
for(int o=0;o<I;o++){
|
||||
if(fabsf(gate[(size_t)c*I+o]-rg[o])>1e-3f*(fabsf(rg[o])+1e-3f)||
|
||||
fabsf(up[(size_t)c*I+o]-ru[o])>1e-3f*(fabsf(ru[o])+1e-3f)) bad++;
|
||||
}
|
||||
cpu_gemv_g4(hd[c],hds[c],I,D,cgs,(float*)gate+(size_t)c*I,ry);
|
||||
for(int o=0;o<D;o++)
|
||||
if(fabsf(y[(size_t)c*D+o]-ry[o])>1e-3f*(fabsf(ry[o])+1e-3f)) bad++;
|
||||
}
|
||||
for(int c=0;c<COUNT;c++){ cudaFree(qg[c]);cudaFree(qu[c]);cudaFree(qd[c]);
|
||||
cudaFree(sg[c]);cudaFree(su[c]);cudaFree(sd[c]);
|
||||
free(hg[c]);free(hu[c]);free(hd[c]);free(hgs[c]);free(hus[c]);free(hds[c]); }
|
||||
cudaFree(ddesc);cudaFree(xs);cudaFree(gate);cudaFree(up);cudaFree(y);
|
||||
}
|
||||
printf("grouped-g4 oracle: %d trials x %d experts (gs=64 + tail + per-row member), %d mismatches\n",
|
||||
trials,COUNT,bad);
|
||||
if(bad){ printf("FAIL\n"); return 1; }
|
||||
printf("OK\n"); return 0;
|
||||
}
|
||||
@@ -18,7 +18,7 @@
|
||||
* index, a wrong group boundary or a swapped nibble cannot hide under it —
|
||||
* those are O(1) relative errors, not O(1e-6). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static uint32_t rng_state=0xC0FFEEu;
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@
|
||||
* (sign-trick kernels must treat |−128| as 128 unsigned, not saturate to 127),
|
||||
* and random data at qrow_i8's contract (|x| <= 127, w full int8 range). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static uint32_t rng_state=0x12345678u;
|
||||
|
||||
@@ -0,0 +1,257 @@
|
||||
"""Inefficiency / regression tests for the colibri engine (tiny model, asserted).
|
||||
|
||||
These run against the bundled glm_tiny model (~0.6 MB resident, ~0.1s/run) and
|
||||
gate CI: a regression here means something broke. They run on a plain CPU-only
|
||||
`glm.exe` build; the CUDA_* tests auto-skip if the engine wasn't built with
|
||||
CUDA_DLL=1 (see tests/README_efficiency.md for the build command).
|
||||
|
||||
The signals under test, and the inefficiency each catches:
|
||||
- tok/s floor : a throughput regression (broken build / bad config)
|
||||
- profile phases sum : telemetry accounting bug (other balloons)
|
||||
- disk-wait not dominant: a tiny resident model should never be I/O-bound
|
||||
- CPU determinism : greedy decode is reproducible (no stray RNG/threading)
|
||||
- CUDA init path : COLI_CUDA=1 initializes and does not silently exit 2
|
||||
- CUDA dense uses VRAM : CUDA_DENSE=1 actually uploads tensors (no silent CPU fallback)
|
||||
- CPU vs CUDA TF-match : identical weights+inputs → identical argmax (kernel bug guard)
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from tools.efficiency import (
|
||||
parse_run, run_engine, disk_wait_share, tf_agreement,
|
||||
TINY_TOK_S_FLOOR, MAX_DISK_WAIT_SHARE, MIN_CPU_CUDA_AGREEMENT,
|
||||
)
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
C_DIR = HERE.parent
|
||||
ENGINE = C_DIR / "glm.exe"
|
||||
TINY = C_DIR / "glm_tiny"
|
||||
|
||||
|
||||
def _engine_present() -> bool:
|
||||
"""True iff BOTH the built engine AND the tiny fixture are available.
|
||||
|
||||
These tests need glm.exe (a build artifact) AND glm_tiny/ (a generated
|
||||
fixture, gitignored). CI runs `make check` = "dependency-free tests, no
|
||||
model downloads" (workflow .github/workflows/check.yml, by design #140), so
|
||||
neither is present there and these tests must SKIP rather than fail. They
|
||||
run locally after `make glm.exe` (glm_tiny ships alongside the source, or
|
||||
is regenerated by tools/make_glm_oracle.py).
|
||||
"""
|
||||
return ENGINE.exists() and (TINY / "config.json").exists()
|
||||
|
||||
|
||||
def _skip_reason() -> str:
|
||||
"""Name exactly which prerequisite is missing, so the skip is actionable."""
|
||||
if not ENGINE.exists():
|
||||
return "glm.exe not built (run: make glm.exe)"
|
||||
if not (TINY / "config.json").exists():
|
||||
return "glm_tiny fixture absent (gitignored; ship it locally or run tools/make_glm_oracle.py)"
|
||||
return ""
|
||||
|
||||
|
||||
def _cuda_available() -> bool:
|
||||
"""True iff the engine binary has the CUDA loader compiled in AND the DLL is present.
|
||||
|
||||
The host is built with -DCOLI_CUDA only when CUDA_DLL=1 (Makefile). A binary
|
||||
built without it embeds the string "this binary is CPU-only; rebuild" and
|
||||
exits 2 on any CUDA env var — so we detect the CPU-only build by scanning
|
||||
the binary for that marker (avoids a slow ldd/strings on every import; we
|
||||
only read enough to find it). On Windows the DLL is also required
|
||||
(backend_loader.c loads it at runtime); on Linux it's direct-linked.
|
||||
"""
|
||||
if not _engine_present():
|
||||
return False
|
||||
try:
|
||||
# Read once; the marker is near the read-only string table. 256 KB is
|
||||
# plenty for this string and avoids loading a 1 MB binary into memory.
|
||||
with open(ENGINE, "rb") as f:
|
||||
blob = f.read(2 * 1024 * 1024)
|
||||
if b"this binary is CPU-only" in blob:
|
||||
return False # CPU-only build: COLI_CUDA=1 would exit 2.
|
||||
except OSError:
|
||||
pass
|
||||
# Windows: DLL required at runtime. Linux: direct-linked (no DLL).
|
||||
if os.name == "nt":
|
||||
return (C_DIR / "coli_cuda.dll").exists()
|
||||
import subprocess
|
||||
try:
|
||||
out = subprocess.run(["ldd", str(ENGINE)], capture_output=True, text=True)
|
||||
return "libcudart" in out.stdout
|
||||
except (FileNotFoundError, OSError):
|
||||
return False
|
||||
|
||||
|
||||
@unittest.skipUnless(_engine_present(), _skip_reason() or "glm.exe + glm_tiny required")
|
||||
class TinyEfficiencyTest(unittest.TestCase):
|
||||
"""Asserted regression tests on the resident tiny model. Gates CI."""
|
||||
|
||||
def _run(self, **overlay):
|
||||
return run_engine(overlay, engine=str(ENGINE), snap=str(TINY))[0]
|
||||
|
||||
# -- telemetry contract ---------------------------------------------------
|
||||
|
||||
def test_telemetry_parses(self):
|
||||
"""A REPLAY run must emit the throughput + PROFILE lines the suite keys on.
|
||||
|
||||
If this fails, either the engine changed its output format (update the
|
||||
parsers in tools/efficiency.py) or the run crashed early."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
self.assertIn("tok_s", t["parsed"], f"missing tok/s line:\n{t['stderr']}")
|
||||
self.assertIn("profile", t["parsed"], f"missing PROFILE line:\n{t['stderr']}")
|
||||
self.assertIn("hit_pct", t["parsed"], f"missing expert hit line:\n{t['stderr']}")
|
||||
self.assertIsNotNone(t["tok_s"])
|
||||
self.assertIsNotNone(t["profile"])
|
||||
|
||||
# -- throughput floor -----------------------------------------------------
|
||||
|
||||
def test_tiny_tok_s_floor(self):
|
||||
"""Tiny decode must beat TINY_TOK_S_FLOOR.
|
||||
|
||||
The tiny model is fully resident and runs ~200 tok/s; the 20 tok/s
|
||||
default floor is a 10x margin that catches broken builds or a pathological
|
||||
config cascade (the cap=1 trap from ISSUE_new_model_resource_regression.md)
|
||||
without flapping on machine noise."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="8")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
self.assertGreaterEqual(
|
||||
t["tok_s"], TINY_TOK_S_FLOOR,
|
||||
f"tok/s {t['tok_s']:.1f} below floor {TINY_TOK_S_FLOOR} "
|
||||
f"(regression, or a config cascade starving the cache)",
|
||||
)
|
||||
|
||||
# -- accounting sanity ----------------------------------------------------
|
||||
|
||||
def test_profile_phases_present_and_nonneg(self):
|
||||
"""Every PROFILE phase must be present and non-negative.
|
||||
|
||||
`other` can go slightly negative from timer overhead (the engine allows
|
||||
it), but a large negative means the timers are double-counting."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="4")
|
||||
p = t["profile"]
|
||||
for phase in ("disk", "expert_matmul", "attention", "lm_head"):
|
||||
self.assertGreaterEqual(p[phase], -0.01, f"{phase} went negative: {p}")
|
||||
# 'other' is a residual; allow a small negative from timer overlap.
|
||||
self.assertGreaterEqual(p["other"], -0.05, f"other too negative (double-count): {p}")
|
||||
|
||||
# -- disk-wait not dominant on a resident model ---------------------------
|
||||
|
||||
def test_disk_wait_not_dominant(self):
|
||||
"""A fully-resident tiny model must NOT be I/O-bound.
|
||||
|
||||
Everything fits in RAM; the expert-disk wait share should be ~0. If it
|
||||
exceeds MAX_DISK_WAIT_SHARE, the cache/I/O path regressed — on a real
|
||||
model this same regression would make decode I/O-bound (the exact
|
||||
failure mode the [PROF] verdict flags)."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="8", PROF="1")
|
||||
self.assertIn("time_shares", t["parsed"], f"missing [PROF] time shares:\n{t['stderr']}")
|
||||
share = disk_wait_share(t)
|
||||
self.assertIsNotNone(share)
|
||||
self.assertLess(
|
||||
share, MAX_DISK_WAIT_SHARE,
|
||||
f"expert-I/O share {share:.0%} on a resident model — I/O path regressed",
|
||||
)
|
||||
|
||||
# -- determinism ----------------------------------------------------------
|
||||
|
||||
def test_cpu_vs_cpu_determinism(self):
|
||||
"""Two greedy REPLAY runs with the same seed produce identical telemetry.
|
||||
|
||||
TEMP=0 = greedy (no sampling), so tok/s and hit-rate must be reproducible.
|
||||
A drift here means non-determinism crept into the decode path (stray
|
||||
threading, uninitialized state) — which on a real model would make A/B
|
||||
comparisons meaningless."""
|
||||
a = self._run(REPLAY="1", TEMP="0", NGEN="8", SEED="1")
|
||||
b = self._run(REPLAY="1", TEMP="0", NGEN="8", SEED="1")
|
||||
self.assertEqual(a["returncode"], 0)
|
||||
self.assertEqual(b["returncode"], 0)
|
||||
self.assertEqual(a["hit_pct"], b["hit_pct"], "greedy hit-rate drifted between runs")
|
||||
# tok/s within 25% — exact equality is too strict across scheduler noise.
|
||||
self.assertLess(abs(a["tok_s"] - b["tok_s"]) / max(a["tok_s"], b["tok_s"]), 0.25)
|
||||
|
||||
|
||||
@unittest.skipUnless(_cuda_available(),
|
||||
_skip_reason() or "CUDA build not present (run: make clean && make glm.exe CUDA_DLL=1 && make cuda-dll)")
|
||||
class TinyCudaEfficiencyTest(unittest.TestCase):
|
||||
"""CUDA-path regression tests on the tiny model. Skip unless CUDA built.
|
||||
|
||||
glm_tiny is small and fully resident, so CUDA here is fast and exercises the
|
||||
real GPU code path (init, dense upload, kernel correctness) without the
|
||||
long load time or memory pressure of the full model. These guard the
|
||||
silent-failure modes that are otherwise invisible:
|
||||
- COLI_CUDA=1 silently falling back to CPU (loader/DLL missing)
|
||||
- CUDA_DENSE=1 uploading nothing
|
||||
- a CUDA kernel producing different argmax than CPU on identical inputs
|
||||
"""
|
||||
|
||||
def _run(self, **overlay):
|
||||
return run_engine(overlay, engine=str(ENGINE), snap=str(TINY))[0]
|
||||
|
||||
def test_cuda_init_path(self):
|
||||
"""COLI_CUDA=1 must initialize the device and NOT exit 2.
|
||||
|
||||
Exit 2 is the engine's "requested backend is unavailable" path
|
||||
(glm.c: g_cuda_enabled check). A clean init prints the [CUDA] device
|
||||
banner to stderr. If this fails, the DLL is broken or the loader can't
|
||||
resolve symbols (ABI drift between backend_cuda.h and the dll)."""
|
||||
t = self._run(COLI_CUDA="1", COLI_GPU="0", REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertNotEqual(t["returncode"], 2,
|
||||
f"engine refused CUDA backend:\n{t['stderr']}")
|
||||
self.assertTrue(t["cuda"]["enabled"],
|
||||
f"no [CUDA] device banner on stderr:\n{t['stderr']}")
|
||||
|
||||
def test_cuda_dense_uses_vram(self):
|
||||
"""CUDA_DENSE=1 must actually upload dense tensors to VRAM.
|
||||
|
||||
The minimal GPU-exercising config (per backend_loader.c analysis):
|
||||
COLI_CUDA=1 + CUDA_DENSE=1. Without CUDA_DENSE the dense path stays on
|
||||
CPU and [CUDA] resident set reports 0 tensors — a silent no-op. This
|
||||
catches that regression: after the run, resident_tensors > 0."""
|
||||
t = self._run(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1",
|
||||
REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
rt = t["cuda"]["resident_tensors"]
|
||||
self.assertIsNotNone(rt, f"no [CUDA] resident set line:\n{t['stderr']}")
|
||||
self.assertGreater(rt, 0,
|
||||
f"CUDA_DENSE=1 but {rt} tensors resident — silent CPU fallback")
|
||||
|
||||
def test_cpu_vs_cuda_tf_match(self):
|
||||
"""CPU and CUDA teacher-forcing must agree on most positions (DIRECTLY).
|
||||
|
||||
Both paths prefill the SAME oracle sequence on the SAME weights, so their
|
||||
argmaxes should match position-for-position — but not exactly: the two
|
||||
backends accumulate dot-products in different orders (x86 SIMD vs CUDA
|
||||
kernel), so a few near-tied logits flip. That divergence is expected
|
||||
numeric behavior, not a kernel bug. A *catastrophic* kernel regression
|
||||
(wrong GEMM, wrong scale, wrong fmt) would collapse agreement toward
|
||||
random (~1/vocab = ~4%); the MIN_CPU_CUDA_AGREEMENT floor (default 70%)
|
||||
catches that while tolerating harmless drift.
|
||||
|
||||
We compare CPU-vs-CUDA *directly* (not via the oracle match-counts),
|
||||
because both backends differ from the oracle at different positions and
|
||||
the summary line can't tell "CPU≠CUDA" from "CPU≠oracle"."""
|
||||
import json
|
||||
ref = json.loads((C_DIR / "ref_glm.json").read_text())
|
||||
oracle = ref["tf_pred"]
|
||||
|
||||
cpu = self._run(TF="1", TEMP="0")
|
||||
cuda = self._run(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1", TF="1", TEMP="0")
|
||||
self.assertEqual(cpu["returncode"], 0, f"CPU TF run failed:\n{cpu['stderr']}")
|
||||
self.assertEqual(cuda["returncode"], 0, f"CUDA TF run failed:\n{cuda['stderr']}")
|
||||
self.assertIn("tf_mismatches", cpu["parsed"], "CPU run missing per-position mismatches")
|
||||
self.assertIn("tf_mismatches", cuda["parsed"], "CUDA run missing per-position mismatches")
|
||||
|
||||
agree, diff = tf_agreement(cpu, cuda, oracle)
|
||||
self.assertGreaterEqual(
|
||||
agree, MIN_CPU_CUDA_AGREEMENT,
|
||||
f"CPU-vs-CUDA argmax agreement {agree:.0%} below floor "
|
||||
f"{MIN_CPU_CUDA_AGREEMENT:.0%} — CUDA kernel likely regressed. "
|
||||
f"Differing positions: {diff[:10]}{'...' if len(diff)>10 else ''}",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,147 @@
|
||||
/* int3-g64 (fmt=5) tests: pack layout, dequant round-trip vs plain-C reference,
|
||||
* matmul_i3 (NEON + scalar tail) vs reference dequant-matmul, per-row helpers,
|
||||
* the .qs-size format tag, and the quality claim in miniature (per-group int3
|
||||
* beats per-row int4 on rows with outliers — the #132 result this format ships). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <math.h>
|
||||
|
||||
static int fails = 0;
|
||||
#define CHECK(c) do{ if(!(c)){ printf("FAIL %s:%d: %s\n", __FILE__, __LINE__, #c); fails++; } }while(0)
|
||||
|
||||
static uint64_t rng = 0x9E3779B97F4A7C15ull;
|
||||
static float rndf(void){ rng ^= rng << 13; rng ^= rng >> 7; rng ^= rng << 17;
|
||||
return ((int64_t)(rng & 0xFFFFF) - 0x80000) / (float)0x80000; }
|
||||
|
||||
/* reference: quantize like pack_int3_g64 but keep dequantized f32 (mirrors
|
||||
* quant_ablation._quant_last_dim(bits=3, group=64)) */
|
||||
static void ref_i3_dequant(const float *w, float *dq, int O, int I){
|
||||
int64_t ng=i3_groups(I);
|
||||
for(int o=0;o<O;o++) for(int64_t g=0;g<ng;g++){
|
||||
int base=(int)(g*I3_GROUP), n=I-base<I3_GROUP?I-base:I3_GROUP;
|
||||
float amax=0; for(int k=0;k<n;k++){ float a=fabsf(w[(int64_t)o*I+base+k]); if(a>amax)amax=a; }
|
||||
float s=amax/3.f; if(s<1e-8f)s=1e-8f;
|
||||
for(int k=0;k<n;k++){
|
||||
int v=(int)lrintf(w[(int64_t)o*I+base+k]/s); if(v>3)v=3; if(v<-4)v=-4;
|
||||
dq[(int64_t)o*I+base+k]=(float)v*s;
|
||||
}
|
||||
}
|
||||
}
|
||||
static void unpack_i3(const uint8_t *q3, const float *s, float *dq, int O, int I){
|
||||
int64_t ng=i3_groups(I), rb=i3_rowbytes(I);
|
||||
for(int o=0;o<O;o++) for(int64_t g=0;g<ng;g++){
|
||||
const uint8_t *lo=q3+(int64_t)o*rb+g*I3_GBYTES, *hi=lo+16;
|
||||
int base=(int)(g*I3_GROUP), n=I-base<I3_GROUP?I-base:I3_GROUP;
|
||||
for(int k=0;k<n;k++){
|
||||
unsigned u=((lo[k>>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2);
|
||||
dq[(int64_t)o*I+base+k]=(float)((int)u-4)*s[(int64_t)o*ng+g];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int main(void){
|
||||
const int Is[]={64,128,192,100,65,7168}; /* incl. short tail groups and one real GLM dim */
|
||||
enum { O=7, MAXI=7168 };
|
||||
static float w[(int64_t)O*MAXI], dq_ref[(int64_t)O*MAXI], dq_pk[(int64_t)O*MAXI];
|
||||
static float x[4*MAXI], y_ref[4*O], y_ker[4*O];
|
||||
static uint8_t q3[(int64_t)O*(MAXI/64+1)*24];
|
||||
static float sc[(int64_t)O*(MAXI/64+1)];
|
||||
|
||||
for(unsigned c=0;c<sizeof Is/sizeof *Is;c++){
|
||||
int I=Is[c];
|
||||
for(int64_t i=0;i<(int64_t)O*I;i++) w[i]=rndf()*0.05f;
|
||||
w[3]=1.7f; w[(int64_t)2*I+5]=-2.2f; /* outliers */
|
||||
|
||||
/* 1. pack -> unpack == reference quantize-dequantize, bit for bit */
|
||||
pack_int3_g64(w, q3, sc, O, I);
|
||||
ref_i3_dequant(w, dq_ref, O, I);
|
||||
unpack_i3(q3, sc, dq_pk, O, I);
|
||||
int bad=0;
|
||||
for(int64_t i=0;i<(int64_t)O*I;i++) if(dq_pk[i]!=dq_ref[i]) bad++;
|
||||
CHECK(bad==0);
|
||||
|
||||
/* 2. matmul_i3 == matmul over the dequantized reference (fp tolerance:
|
||||
* NEON fma order differs from the scalar reference loop) */
|
||||
for(int S=1;S<=4;S+=3){
|
||||
for(int64_t i=0;i<(int64_t)S*I;i++) x[i]=rndf();
|
||||
matmul_i3(y_ker, x, q3, sc, S, I, O);
|
||||
for(int s=0;s<S;s++) for(int o=0;o<O;o++){
|
||||
double a=0; for(int i=0;i<I;i++) a+=(double)dq_ref[(int64_t)o*I+i]*x[(int64_t)s*I+i];
|
||||
y_ref[s*O+o]=(float)a;
|
||||
}
|
||||
for(int i=0;i<S*O;i++){
|
||||
float d=fabsf(y_ker[i]-y_ref[i]), m=fabsf(y_ref[i])>1?fabsf(y_ref[i]):1;
|
||||
if(d/m>2e-4f){ CHECK(!"matmul_i3 mismatch"); break; }
|
||||
}
|
||||
}
|
||||
|
||||
/* 3. QT plumbing: qt_alloc(bits=3) -> qt_fill -> matmul_qt & qt_bytes & helpers */
|
||||
QT t; qt_alloc(&t, O, I, 3);
|
||||
CHECK(t.fmt==5);
|
||||
qt_fill(&t, w, 3);
|
||||
CHECK(qt_bytes(&t)==(int64_t)O*i3_rowbytes(I)+(int64_t)O*i3_groups(I)*4);
|
||||
matmul_qt(y_ker, x, &t, 1);
|
||||
for(int o=0;o<O;o++){
|
||||
double a=0; for(int i=0;i<I;i++) a+=(double)dq_ref[(int64_t)o*I+i]*x[i];
|
||||
float d=fabsf(y_ker[o]-(float)a), m=fabsf((float)a)>1?fabsf((float)a):1;
|
||||
CHECK(d/m<=2e-4f);
|
||||
}
|
||||
float acc[MAXI]; memset(acc,0,I*sizeof(float));
|
||||
qt_addrow(&t, 2, 0.5f, acc);
|
||||
for(int i=0;i<I;i++) CHECK(fabsf(acc[i]-0.5f*dq_ref[(int64_t)2*I+i])<=1e-6f);
|
||||
float yr[3];
|
||||
qt_matvec_rows(&t, 1, 3, x, yr);
|
||||
for(int j=0;j<3;j++){
|
||||
double a=0; for(int i=0;i<I;i++) a+=(double)dq_ref[(int64_t)(1+j)*I+i]*x[i];
|
||||
float d=fabsf(yr[j]-(float)a), m=fabsf((float)a)>1?fabsf((float)a):1;
|
||||
CHECK(d/m<=2e-4f);
|
||||
}
|
||||
|
||||
/* 4. format resolution through the #413 gate: fmt=5 is tagged by its distinct
|
||||
* WEIGHT byte count. int3-g64 and grouped-int4-at-gs=64 carry the SAME scale
|
||||
* cardinality O*ceil(I/64), so the pair (weight bytes, scale bytes) must
|
||||
* disambiguate: same scales, int4 weights -> fmt=4/gs=64; int3 weights -> fmt=5.
|
||||
* Only well-posed for I > 256: below that, O row scales legitimately match a
|
||||
* 1-group grouped layout too (detect_group_size probes gs up to 256), so
|
||||
* per-row vs grouped is not distinguishable from byte counts alone. */
|
||||
if(I>256){ int gs=-1;
|
||||
int64_t ns_g64=(int64_t)O*i3_groups(I)*4, ns_row=(int64_t)O*4;
|
||||
CHECK(qt_resolve_fmt("t.i3", O, I, (int64_t)O*i3_rowbytes(I), ns_g64, &gs)==5);
|
||||
CHECK(gs==0);
|
||||
CHECK(qt_resolve_fmt("t.i8", O, I, (int64_t)O*I, ns_row, &gs)==1);
|
||||
CHECK(qt_resolve_fmt("t.i4", O, I, (int64_t)O*((I+1)/2), ns_row, &gs)==2);
|
||||
CHECK(qt_resolve_fmt("t.i4g", O, I, (int64_t)O*((I+1)/2), ns_g64, &gs)==4);
|
||||
CHECK(gs==64);
|
||||
CHECK(qt_resolve_fmt("t.i2", O, I, (int64_t)O*((I+3)/4), ns_row, &gs)==3); }
|
||||
free(t.q4); free(t.s);
|
||||
}
|
||||
|
||||
/* 5. quality in miniature: on rows with outliers, per-group int3 must beat
|
||||
* per-row int4 on reconstruction RMS (the #132 finding this format ships). */
|
||||
{
|
||||
int I=1024;
|
||||
for(int64_t i=0;i<(int64_t)O*I;i++) w[i]=rndf()*0.02f;
|
||||
for(int o=0;o<O;o++) w[(int64_t)o*I+(o*37)%I]=1.5f; /* one outlier per row */
|
||||
ref_i3_dequant(w, dq_ref, O, I);
|
||||
QT t4; qt_alloc(&t4, O, I, 4); qt_fill(&t4, w, 4);
|
||||
double e3=0, e4=0;
|
||||
for(int o=0;o<O;o++) for(int i=0;i<I;i++){
|
||||
float w4; { const uint8_t *q=t4.q4+(int64_t)o*((I+1)/2); uint8_t b=q[i>>1];
|
||||
int v=(i&1)?((int)(b>>4)-8):((int)(b&0xF)-8); w4=(float)v*t4.s[o]; }
|
||||
double d3=w[(int64_t)o*I+i]-dq_ref[(int64_t)o*I+i], d4=w[(int64_t)o*I+i]-w4;
|
||||
e3+=d3*d3; e4+=d4*d4;
|
||||
}
|
||||
CHECK(e3 < e4);
|
||||
printf(" outlier-rows RMS: int3-g64 %.3e < int4-row %.3e (ratio %.2f)\n",
|
||||
sqrt(e3/((double)O*I)), sqrt(e4/((double)O*I)), sqrt(e4/e3));
|
||||
free(t4.q4); free(t4.s);
|
||||
}
|
||||
|
||||
if(fails){ printf("int3-g64 tests: %d FAILED\n", fails); return 1; }
|
||||
printf("int3-g64 tests: ok\n");
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
"""quant_int3_g64 (tools/convert_fp8_to_int4.py): pack layout + round-trip.
|
||||
|
||||
Decodes the packed bytes with an independent NumPy decoder implementing the
|
||||
fmt=5 spec (16B low plane / 8B high plane per 64-group, v+4, per-group f32
|
||||
scale) and checks the dequantized result equals the reference
|
||||
quantize-dequantize (same math as quant_ablation._quant_last_dim(3, 64)).
|
||||
The C side of the same layout is covered by tests/test_int3.c.
|
||||
"""
|
||||
import os, sys, unittest
|
||||
try:
|
||||
import numpy as np
|
||||
except ImportError:
|
||||
raise unittest.SkipTest("numpy not installed")
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||
from convert_fp8_to_int4 import quant_int3_g64
|
||||
|
||||
|
||||
def decode(packed, scales, O, I, group=64):
|
||||
ng = (I + group - 1) // group
|
||||
b = packed.reshape(O, ng, 24)
|
||||
lo, hi = b[:, :, :16], b[:, :, 16:]
|
||||
k = np.arange(group)
|
||||
lov = (lo[:, :, k >> 2] >> ((k & 3) * 2)[None, None, :]) & 3
|
||||
hiv = (hi[:, :, k >> 3] >> (k & 7)[None, None, :]) & 1
|
||||
v = (lov | (hiv << 2)).astype(np.int64) - 4
|
||||
dq = v.astype(np.float64) * scales.reshape(O, ng, 1).astype(np.float64)
|
||||
return dq.reshape(O, ng * group)[:, :I]
|
||||
|
||||
|
||||
def reference(w, group=64):
|
||||
"""same math as quant_int3_g64 (which works in f32), replayed exactly, then
|
||||
dequantized in f64 so it matches decode() bit for bit"""
|
||||
O, I = w.shape
|
||||
ng = (I + group - 1) // group
|
||||
pad = ng * group - I
|
||||
wp = np.pad(w, ((0, 0), (0, pad))) if pad else w
|
||||
g = wp.reshape(O, ng, group)
|
||||
s = np.maximum(np.abs(g).max(axis=2, keepdims=True) / 3.0, 1e-8).astype(np.float32)
|
||||
q = np.clip(np.rint(g / s), -4, 3).astype(np.int64)
|
||||
return (q.astype(np.float64) * s.astype(np.float64)).reshape(O, ng * group)[:, :I]
|
||||
|
||||
|
||||
class Int3ConvertTest(unittest.TestCase):
|
||||
def test_round_trip(self):
|
||||
rng = np.random.default_rng(7)
|
||||
for I in (64, 128, 100, 65, 7168):
|
||||
w = (rng.standard_normal((5, I)) * 0.05).astype(np.float32)
|
||||
w[0, 3] = 1.7; w[2, min(5, I - 1)] = -2.2
|
||||
packed, scales = quant_int3_g64(w)
|
||||
ng = (I + 63) // 64
|
||||
self.assertEqual(packed.size, 5 * ng * 24)
|
||||
self.assertEqual(scales.size, 5 * ng)
|
||||
np.testing.assert_allclose(decode(packed, scales, 5, I),
|
||||
reference(w), rtol=0, atol=0)
|
||||
|
||||
def test_outliers_beat_row_int4(self):
|
||||
rng = np.random.default_rng(11)
|
||||
w = (rng.standard_normal((8, 1024)) * 0.02).astype(np.float32)
|
||||
for o in range(8): w[o, (o * 37) % 1024] = 1.5
|
||||
packed, scales = quant_int3_g64(w)
|
||||
e3 = float(((decode(packed, scales, 8, 1024) - w) ** 2).mean())
|
||||
s4 = np.maximum(np.abs(w).max(axis=1, keepdims=True) / 7.0, 1e-8)
|
||||
w4 = np.clip(np.rint(w / s4), -8, 7) * s4
|
||||
e4 = float(((w4 - w) ** 2).mean())
|
||||
self.assertLess(e3, e4)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,103 @@
|
||||
/* Loader-seam test for fmt=5: writes a real .safetensors file containing an
|
||||
* int3-g64 tensor (U8 payload + per-GROUP .qs) next to an int4 control tensor
|
||||
* (per-row .qs), indexes it with st_init, loads both through qt_from_disk, and
|
||||
* checks the byte-count/.qs-size format inference picks fmt=5 vs fmt=2 correctly
|
||||
* and the loaded weights dequantize identically to pack_int3_g64's output. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
static int fails = 0;
|
||||
#define CHECK(c) do{ if(!(c)){ printf("FAIL %s:%d: %s\n", __FILE__, __LINE__, #c); fails++; } }while(0)
|
||||
|
||||
static uint64_t rng = 0xA5A5A5A55A5A5A5Aull;
|
||||
static float rndf(void){ rng ^= rng << 13; rng ^= rng >> 7; rng ^= rng << 17;
|
||||
return ((int64_t)(rng & 0xFFFFF) - 0x80000) / (float)0x80000; }
|
||||
|
||||
static void deq4(const QT *t, float *dq){
|
||||
for(int o=0;o<t->O;o++) for(int i=0;i<t->I;i++){
|
||||
if(t->fmt==5){
|
||||
int64_t g=i/I3_GROUP; const uint8_t *lo=t->q4+(int64_t)o*i3_rowbytes(t->I)+g*I3_GBYTES, *hi=lo+16;
|
||||
int k=i%I3_GROUP;
|
||||
unsigned u=((lo[k>>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2);
|
||||
dq[(int64_t)o*t->I+i]=(float)((int)u-4)*t->s[(int64_t)o*i3_groups(t->I)+g];
|
||||
} else { /* fmt2 */
|
||||
uint8_t b=t->q4[(int64_t)o*((t->I+1)/2)+(i>>1)];
|
||||
int v=(i&1)?((int)(b>>4)-8):((int)(b&0xF)-8);
|
||||
dq[(int64_t)o*t->I+i]=(float)v*t->s[o];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int main(void){
|
||||
enum { O=5, I=320 }; /* 5 groups per row; I > 256 so the per-row int4
|
||||
* control stays fmt=2 (detect_group_size probes
|
||||
* gs up to 256: any smaller I would make O row
|
||||
* scales match a legitimate 1-group layout) */
|
||||
int64_t ng=i3_groups(I), rb=i3_rowbytes(I);
|
||||
static float w[O*I];
|
||||
for(int i=0;i<O*I;i++) w[i]=rndf()*0.05f;
|
||||
w[7]=1.9f;
|
||||
|
||||
static uint8_t q3[O*(I/64)*24]; static float s3[O*(I/64)];
|
||||
pack_int3_g64(w, q3, s3, O, I);
|
||||
static uint8_t q4b[O*((I+1)/2)]; static float s4[O];
|
||||
pack_int4(w, q4b, s4, O, I, 4);
|
||||
|
||||
/* write a minimal single-shard safetensors file */
|
||||
const char *dir="tests/tmp_int3_snap";
|
||||
#ifdef _WIN32
|
||||
mkdir(dir);
|
||||
#else
|
||||
mkdir(dir, 0755);
|
||||
#endif
|
||||
char path[256]; snprintf(path,sizeof path,"%s/model.safetensors",dir);
|
||||
int64_t nb3=(int64_t)O*rb, ns3=(int64_t)O*ng*4, nb4=(int64_t)O*((I+1)/2), ns4=(int64_t)O*4;
|
||||
char hdr[1024];
|
||||
int hl=snprintf(hdr,sizeof hdr,
|
||||
"{\"w3\":{\"dtype\":\"U8\",\"shape\":[%lld],\"data_offsets\":[0,%lld]},"
|
||||
"\"w3.qs\":{\"dtype\":\"F32\",\"shape\":[%lld],\"data_offsets\":[%lld,%lld]},"
|
||||
"\"w4\":{\"dtype\":\"U8\",\"shape\":[%lld],\"data_offsets\":[%lld,%lld]},"
|
||||
"\"w4.qs\":{\"dtype\":\"F32\",\"shape\":[%lld],\"data_offsets\":[%lld,%lld]}}",
|
||||
(long long)nb3,(long long)nb3,
|
||||
(long long)(O*ng),(long long)nb3,(long long)(nb3+ns3),
|
||||
(long long)nb4,(long long)(nb3+ns3),(long long)(nb3+ns3+nb4),
|
||||
(long long)O,(long long)(nb3+ns3+nb4),(long long)(nb3+ns3+nb4+ns4));
|
||||
FILE *f=fopen(path,"wb");
|
||||
if(!f){ printf("FAIL: cannot create %s (run from c/, like tools/run_tests.py does)\n", path); return 1; }
|
||||
uint64_t hlen=(uint64_t)hl;
|
||||
fwrite(&hlen,8,1,f); fwrite(hdr,1,hl,f);
|
||||
fwrite(q3,1,(size_t)nb3,f); fwrite(s3,1,(size_t)ns3,f);
|
||||
fwrite(q4b,1,(size_t)nb4,f); fwrite(s4,1,(size_t)ns4,f);
|
||||
fclose(f);
|
||||
|
||||
static Model gm; /* only gm.S is used by qt_from_disk */
|
||||
st_init(&gm.S, dir);
|
||||
|
||||
QT t3; memset(&t3,0,sizeof t3);
|
||||
qt_from_disk(&gm,"w3",O,I,8,0,&t3);
|
||||
CHECK(t3.fmt==5);
|
||||
static float dq_load[O*I], dq_ref[O*I];
|
||||
deq4(&t3,dq_load);
|
||||
QT tr={.fmt=5,.q4=q3,.s=s3,.O=O,.I=I};
|
||||
deq4(&tr,dq_ref);
|
||||
CHECK(memcmp(dq_load,dq_ref,sizeof dq_ref)==0);
|
||||
|
||||
QT t4; memset(&t4,0,sizeof t4);
|
||||
qt_from_disk(&gm,"w4",O,I,8,0,&t4);
|
||||
CHECK(t4.fmt==2); /* control: row-scale int4 still detected */
|
||||
deq4(&t4,dq_load);
|
||||
QT tr4={.fmt=2,.q4=q4b,.s=s4,.O=O,.I=I};
|
||||
deq4(&tr4,dq_ref);
|
||||
CHECK(memcmp(dq_load,dq_ref,sizeof dq_ref)==0);
|
||||
|
||||
unlink(path); rmdir(dir);
|
||||
if(fails){ printf("int3 loader tests: %d FAILED\n", fails); return 1; }
|
||||
printf("int3 loader tests: ok\n");
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
"""fmt=5 codec oracle (#452 ladder step 2).
|
||||
|
||||
Checks the properties the container, the converter and the decode kernels all
|
||||
depend on: exact byte budget, deterministic encode, decode agreeing with a
|
||||
straight-from-the-spec reader, sign parity closure, and reconstruction quality
|
||||
matching the ablation that chose this scheme (#453).
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
# The runtime path is dependency-free by design and CI keeps it that way, so the
|
||||
# offline-tooling tests skip rather than fail where numpy is absent.
|
||||
np = None
|
||||
P = None
|
||||
try:
|
||||
import numpy as np
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
|
||||
import iq3_pack as P
|
||||
except ImportError: # pragma: no cover - exercised only on dependency-free CI
|
||||
pass
|
||||
|
||||
|
||||
def ref_decode(packed, K):
|
||||
"""Independent reader written straight from the layout comment — deliberately
|
||||
naive and loop-based, so a shared bug in the vectorized path shows up."""
|
||||
g = P.grid()
|
||||
nsb = K // P.QK
|
||||
rows = packed.reshape(-1, nsb * P.BLOCK_BYTES)
|
||||
out = np.zeros((len(rows), K), dtype=np.float32)
|
||||
for r in range(len(rows)):
|
||||
for sb in range(nsb):
|
||||
base = sb * P.BLOCK_BYTES
|
||||
d = float(rows[r, base + 96:base + 98].copy().view(np.float16)[0])
|
||||
for ib in range(P.QK // P.SUB):
|
||||
word = int(np.ascontiguousarray(
|
||||
rows[r, base + 64 + ib * 4:base + 64 + ib * 4 + 4]).view(np.uint32)[0])
|
||||
db = d * (0.5 + ((word >> 28) & 0xF)) * 0.5
|
||||
for l in range(4):
|
||||
seven = (word >> (7 * l)) & 0x7F
|
||||
bits = [(seven >> j) & 1 for j in range(7)]
|
||||
bits.append(sum(bits) & 1) # odd parity closes the 8th
|
||||
idx = int(rows[r, base + ib * 8 + l * 2 + 0])
|
||||
idx2 = int(rows[r, base + ib * 8 + l * 2 + 1])
|
||||
mags = list(g[idx]) + list(g[idx2])
|
||||
for j in range(8):
|
||||
pos = sb * P.QK + ib * P.SUB + l * 8 + j
|
||||
out[r, pos] = mags[j] * db * (-1.0 if bits[j] else 1.0)
|
||||
return out
|
||||
|
||||
|
||||
@unittest.skipIf(P is None, "numpy not available (offline-tooling test)")
|
||||
class TestIq3Pack(unittest.TestCase):
|
||||
def setUp(self):
|
||||
np.random.seed(4242)
|
||||
self.x = (np.random.randn(6, 1024) * 0.05).astype(np.float32)
|
||||
|
||||
def test_byte_budget(self):
|
||||
self.assertEqual(P.BLOCK_BYTES, 98)
|
||||
self.assertAlmostEqual(P.bpw(), 3.0625, places=6)
|
||||
packed = P.encode(self.x)
|
||||
self.assertEqual(packed.shape, (6, 1024 // P.QK * 98))
|
||||
self.assertEqual(packed.dtype, np.uint8)
|
||||
|
||||
def test_encode_is_deterministic(self):
|
||||
self.assertTrue(np.array_equal(P.encode(self.x), P.encode(self.x)))
|
||||
|
||||
def test_decode_matches_spec_reader(self):
|
||||
packed = P.encode(self.x)
|
||||
fast = P.decode(packed, 1024)
|
||||
slow = ref_decode(packed, 1024)
|
||||
self.assertTrue(np.allclose(fast, slow, rtol=1e-6, atol=1e-8),
|
||||
f"max |Δ| = {np.abs(fast - slow).max()}")
|
||||
|
||||
def test_sign_parity_closes(self):
|
||||
"""Every 8-weight block must have an even number of negatives — that is
|
||||
what lets the 8th sign be derived instead of stored."""
|
||||
y = P.decode(P.encode(self.x), 1024)
|
||||
neg = (y < 0).reshape(-1, 8).sum(-1)
|
||||
self.assertTrue(np.all(neg % 2 == 0), "a block stored odd negatives")
|
||||
|
||||
def test_reconstruction_quality(self):
|
||||
y = P.decode(P.encode(self.x), 1024)
|
||||
rel = np.sqrt(((y - self.x) ** 2).mean()) / np.sqrt((self.x ** 2).mean())
|
||||
# the torch model that won the #453 A/B measures ~0.195 on this input class
|
||||
self.assertLess(rel, 0.25, f"rel-RMSE {rel:.4f} — worse than the chosen scheme")
|
||||
self.assertGreater(rel, 0.05, f"rel-RMSE {rel:.4f} — implausibly good, check the test")
|
||||
|
||||
def test_shape_guard(self):
|
||||
with self.assertRaises(ValueError):
|
||||
P.encode(np.zeros((2, 300), dtype=np.float32))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -4,7 +4,7 @@
|
||||
* arrays twice on the second call -> allocator abort. No model file needed:
|
||||
* the CPU path of kv_alloc only reads c->n_layers/kv_lora/qk_rope. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
int main(void){
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
#include <assert.h>
|
||||
#include <math.h>
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static int approx1(double x){ return x > 0.999 && x < 1.001; }
|
||||
|
||||
@@ -35,7 +35,7 @@ class MakefilePlatformTests(unittest.TestCase):
|
||||
env["PATH"] = ""
|
||||
|
||||
result = subprocess.run(
|
||||
[MAKE, "--no-print-directory", "-B", "-n", "glm"],
|
||||
[MAKE, "--no-print-directory", "-B", "-n", "colibri"],
|
||||
cwd=C_DIR,
|
||||
env=env,
|
||||
text=True,
|
||||
@@ -43,7 +43,7 @@ class MakefilePlatformTests(unittest.TestCase):
|
||||
check=True,
|
||||
)
|
||||
|
||||
self.assertIn("-o glm.exe", result.stdout)
|
||||
self.assertIn("-o colibri.exe", result.stdout)
|
||||
self.assertIn("-fopenmp", result.stdout)
|
||||
self.assertIn("-static", result.stdout)
|
||||
|
||||
|
||||
@@ -20,8 +20,8 @@ class FakeEngine:
|
||||
self.calls = []
|
||||
|
||||
def generate(self, prompt, maximum, temperature, top_p, on_text, cache_slot=0,
|
||||
cancelled=None):
|
||||
self.calls.append((prompt, maximum, temperature, top_p, cache_slot))
|
||||
cancelled=None, grammar=None):
|
||||
self.calls.append((prompt, maximum, temperature, top_p, cache_slot, grammar))
|
||||
on_text("Hé")
|
||||
on_text("llo")
|
||||
return {"prompt_tokens": 7, "completion_tokens": 2, "length_limited": False}
|
||||
@@ -34,7 +34,7 @@ class BlockingEngine(FakeEngine):
|
||||
self.release = threading.Event()
|
||||
|
||||
def generate(self, prompt, maximum, temperature, top_p, on_text, cache_slot=0,
|
||||
cancelled=None):
|
||||
cancelled=None, grammar=None):
|
||||
self.entered.set()
|
||||
self.release.wait(2)
|
||||
return super().generate(prompt, maximum, temperature, top_p, on_text, cache_slot,
|
||||
@@ -71,11 +71,11 @@ class TemplateTest(unittest.TestCase):
|
||||
|
||||
def test_validates_generation_limits(self):
|
||||
self.assertEqual(generation_options({"max_tokens": 4, "temperature": 0, "top_p": 1}, 8),
|
||||
(4, 0.0, 1.0))
|
||||
(4, 0.0, 1.0, None))
|
||||
# max_tokens above the server cap is clamped, not rejected (#260): OpenAI
|
||||
# clients default to large values; erroring breaks them.
|
||||
self.assertEqual(generation_options({"max_tokens": 9, "temperature": 0, "top_p": 1}, 8),
|
||||
(8, 0.0, 1.0))
|
||||
(8, 0.0, 1.0, None))
|
||||
# non-positive / non-int max_tokens is still a hard error
|
||||
with self.assertRaises(APIError):
|
||||
generation_options({"max_tokens": 0}, 8)
|
||||
@@ -84,7 +84,31 @@ class TemplateTest(unittest.TestCase):
|
||||
with self.assertRaises(APIError):
|
||||
generation_options({"top_p": math.inf}, 8)
|
||||
self.assertEqual(generation_options({"temperature": None, "top_p": None}, 8),
|
||||
(8, 0.7, 0.9))
|
||||
(8, 0.7, 0.9, None))
|
||||
# response_format -> grammar plumbing (draft source, never a constraint)
|
||||
opts = generation_options({"max_tokens": 4, "response_format": {"type": "json_object"}}, 8)
|
||||
self.assertIn("root ::=", opts[3])
|
||||
schema = {"type": "object", "properties": {"a": {"type": "string"}}, "required": ["a"]}
|
||||
opts = generation_options({"max_tokens": 4, "response_format":
|
||||
{"type": "json_schema", "json_schema": {"schema": schema}}}, 8)
|
||||
self.assertEqual(json.loads(opts[3]), schema)
|
||||
opts = generation_options({"max_tokens": 4, "response_format":
|
||||
{"type": "gbnf", "grammar": 'root ::= "x"'}}, 8)
|
||||
self.assertEqual(opts[3], 'root ::= "x"')
|
||||
with self.assertRaises(APIError):
|
||||
generation_options({"response_format": {"type": "yaml"}}, 8)
|
||||
with self.assertRaises(APIError):
|
||||
generation_options({"response_format": {"type": "json_schema", "json_schema": {}}}, 8)
|
||||
with self.assertRaises(APIError): # non-dict response_format
|
||||
generation_options({"response_format": "json"}, 8)
|
||||
with self.assertRaises(APIError): # empty gbnf
|
||||
generation_options({"response_format": {"type": "gbnf", "grammar": " "}}, 8)
|
||||
with self.assertRaises(APIError): # oversized grammar (> 1 MiB pre-check)
|
||||
generation_options({"response_format": {"type": "gbnf", "grammar": "x" * ((1 << 20) + 1)}}, 8)
|
||||
# malformed GBNF passes the gateway by design: the ENGINE fail-softs it
|
||||
# (draft source only — bad grammar costs the speedup, never the request)
|
||||
opts = generation_options({"response_format": {"type": "gbnf", "grammar": "not a grammar ::="}}, 8)
|
||||
self.assertEqual(opts[3], "not a grammar ::=")
|
||||
|
||||
|
||||
class ProtocolTest(unittest.TestCase):
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
/* COLI_PIPE_BLOCK: the pipe pool's condvar waiter must be observably
|
||||
* equivalent to the sched_yield spin it replaces — same bytes land in the
|
||||
* same ws[] slots, and no interleaving loses a wakeup (the worker RELEASE-
|
||||
* stores ready[] BEFORE taking mx to broadcast; the waiter re-checks under
|
||||
* the lock, so a flag set between its fast-path check and the wait cannot
|
||||
* be missed). Both waiters are exercised against the same on-disk fixture,
|
||||
* alternating parked waits (wait issued before the load finishes) with
|
||||
* fast-path waits (load already done), across enough generations to cycle
|
||||
* the pool's gen-tagged cursor.
|
||||
*
|
||||
* Also pins the PIPE_WORKERS => PIPE implication table: fires ONLY when
|
||||
* PIPE is unset in the env AND the platform default left the pipe off AND
|
||||
* PIPE_WORKERS parses positive (PIPE_WORKERS=0/empty/negative must NOT
|
||||
* silently enable a clamped 1-worker pipe). */
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
#define main coli_glm_main_unused
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static int fail(const char *s){ fprintf(stderr,"FAIL: %s\n",s); return 1; }
|
||||
|
||||
enum { NE=8, LAYER=1 }; /* experts 0..NE-1 on one MoE layer */
|
||||
/* per-expert file image: [gate 12][up 12][down 12][gate.qs 12][up.qs 12][down.qs 16] */
|
||||
enum { WB=12, QS_G=12, QS_U=12, QS_D=16, ESZ=3*WB+QS_G+QS_U+QS_D };
|
||||
|
||||
static unsigned char wbyte(int e,int j){ return (unsigned char)(e*31+j+1); }
|
||||
static float scale(int e,int i){ return (float)(e*8+i)+0.5f; }
|
||||
|
||||
#define TMPF "test_pipe_block.tmp"
|
||||
|
||||
static int write_fixture(void){
|
||||
FILE *w=fopen(TMPF,"wb"); if(!w) return fail("create temp");
|
||||
for(int e=0;e<NE;e++){
|
||||
unsigned char img[ESZ];
|
||||
for(int j=0;j<3*WB;j++) img[j]=wbyte(e,j);
|
||||
float sc[(QS_G+QS_U+QS_D)/4];
|
||||
for(int i=0;i<(int)(sizeof(sc)/sizeof(sc[0]));i++) sc[i]=scale(e,i);
|
||||
memcpy(img+3*WB,sc,sizeof(sc));
|
||||
if(fwrite(img,1,ESZ,w)!=ESZ){ fclose(w); return fail("expert fixture write"); }
|
||||
}
|
||||
fclose(w);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int build_fixture(Model *m,int fd){
|
||||
m->c.hidden=4; m->c.moe_inter=3; m->ebits=8;
|
||||
m->S.n=NE*6; m->S.cap=NE*6; m->S.t=calloc(NE*6,sizeof(st_tensor));
|
||||
if(!m->S.t) return fail("tensor metadata allocation");
|
||||
const char *proj[3]={"gate_proj","up_proj","down_proj"};
|
||||
int sbytes[3]={QS_G,QS_U,QS_D};
|
||||
for(int e=0;e<NE;e++){
|
||||
int64_t wo=(int64_t)e*ESZ, so=wo+3*WB;
|
||||
for(int k=0;k<3;k++){
|
||||
char name[300];
|
||||
snprintf(name,sizeof(name),"model.layers.%d.mlp.experts.%d.%s.weight",LAYER,e,proj[k]);
|
||||
m->S.t[e*6+k]=(st_tensor){strdup(name),fd,wo,WB,3,WB}; wo+=WB;
|
||||
size_t n=strlen(name); memcpy(name+n,".qs",4);
|
||||
m->S.t[e*6+3+k]=(st_tensor){strdup(name),fd,so,sbytes[k],2,sbytes[k]/4}; so+=sbytes[k];
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int check_slot(ESlot *s,int e){
|
||||
if(s->eid!=e || s->g.fmt!=1 || s->u.fmt!=1 || s->d.fmt!=1){
|
||||
fprintf(stderr," slot: eid=%d (want %d) fmt g/u/d=%d/%d/%d (want 1/1/1)\n",
|
||||
s->eid,e,s->g.fmt,s->u.fmt,s->d.fmt);
|
||||
return 1;
|
||||
}
|
||||
const unsigned char *g=(const unsigned char*)s->g.q8,
|
||||
*u=(const unsigned char*)s->u.q8,
|
||||
*d=(const unsigned char*)s->d.q8; /* q8 is int8_t; compare raw bytes */
|
||||
for(int j=0;j<WB;j++)
|
||||
if(g[j]!=wbyte(e,j) || u[j]!=wbyte(e,WB+j) || d[j]!=wbyte(e,2*WB+j)){
|
||||
fprintf(stderr," slot e=%d weight byte %d: g=%d/%d u=%d/%d d=%d/%d (got/want)\n",e,j,
|
||||
g[j],wbyte(e,j),u[j],wbyte(e,WB+j),d[j],wbyte(e,2*WB+j));
|
||||
return 1;
|
||||
}
|
||||
for(int i=0;i<3;i++)
|
||||
if(s->g.s[i]!=scale(e,i) || s->u.s[i]!=scale(e,3+i)){
|
||||
fprintf(stderr," slot e=%d scale %d: g=%g/%g u=%g/%g (got/want)\n",e,i,
|
||||
(double)s->g.s[i],(double)scale(e,i),(double)s->u.s[i],(double)scale(e,3+i));
|
||||
return 1;
|
||||
}
|
||||
for(int i=0;i<4;i++)
|
||||
if(s->d.s[i]!=scale(e,6+i)){
|
||||
fprintf(stderr," slot e=%d scale %d: d=%g/%g (got/want)\n",e,i,
|
||||
(double)s->d.s[i],(double)scale(e,6+i));
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int run_generations(Model *m,int block,int gens){
|
||||
g_pipe_block=block;
|
||||
for(int gen=0;gen<gens;gen++){
|
||||
int eids[NE];
|
||||
for(int q=0;q<NE;q++) eids[q]=(gen*3+q)%NE; /* deterministic shuffle across gens */
|
||||
pipe_dispatch(m,LAYER,eids,NE);
|
||||
if(gen%4==0) usleep(300); /* let loads finish → fast-path wait */
|
||||
for(int i=0;i<NE;i++){
|
||||
/* odd gens wait on the LAST-dispatched slot first: with jobs this
|
||||
* small, in-order waits mostly find ready already set — reverse
|
||||
* order is what actually parks the waiter on the condvar. */
|
||||
int q=(gen&1)?NE-1-i:i;
|
||||
pipe_wait(q);
|
||||
if(!atomic_load_explicit(&g_pp.ready[q],memory_order_acquire))
|
||||
return fail(block?"blocking wait returned before ready":"spin wait returned before ready");
|
||||
if(check_slot(&m->ws[q],eids[q])) return fail(block?"slot contents (block)":"slot contents (spin)");
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int test_implication_table(void){
|
||||
struct { const char *pipe_env,*pw_env; int pipe_now,want; } T[]={
|
||||
{NULL,"4",0,1}, /* pool sized, pipe off, PIPE unset → imply */
|
||||
{NULL,"16",0,1},
|
||||
{NULL,"0",0,0}, /* PIPE_WORKERS=0 must NOT enable a clamped pipe */
|
||||
{NULL,"",0,0},
|
||||
{NULL,"-3",0,0},
|
||||
{"0","4",0,0}, /* explicit PIPE=0 always wins */
|
||||
{"1","4",1,0}, /* explicit PIPE=1: nothing to imply */
|
||||
{NULL,"4",1,0}, /* platform default already ON (win32) */
|
||||
{NULL,NULL,0,0},
|
||||
};
|
||||
for(size_t i=0;i<sizeof(T)/sizeof(T[0]);i++)
|
||||
if(pipe_workers_imply_pipe(T[i].pipe_env,T[i].pw_env,T[i].pipe_now)!=T[i].want){
|
||||
fprintf(stderr,"FAIL: implication row %zu (PIPE=%s PIPE_WORKERS=%s pipe_now=%d)\n",
|
||||
i,T[i].pipe_env?T[i].pipe_env:"<unset>",T[i].pw_env?T[i].pw_env:"<unset>",T[i].pipe_now);
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(void){
|
||||
if(test_implication_table()) return 1;
|
||||
|
||||
/* Relative to the CWD, like test_compat_direct's TMPF — NOT "/tmp/...":
|
||||
* the windows job builds native .exe files and "/tmp" is not a Windows
|
||||
* path. fwrite then reopen read-only: Windows compat has pread, not pwrite. */
|
||||
if(write_fixture()) return 1;
|
||||
int fd=open(TMPF,COMPAT_O_RDONLY);
|
||||
if(fd<0) return fail("open temp");
|
||||
|
||||
static Model m; /* zeroed: buffered pread path, no mmap/cuda */
|
||||
if(build_fixture(&m,fd)){ close(fd); remove(TMPF); return 1; }
|
||||
|
||||
g_pipe=1; g_pipe_nw=4;
|
||||
pipe_init(&m);
|
||||
|
||||
/* spin waiter first (control), then the condvar waiter under the same
|
||||
* dispatch pattern; 200 generations each cycles the gen-tagged cursor
|
||||
* and alternates parked/fast-path waits. */
|
||||
if(run_generations(&m,0,200) || run_generations(&m,1,200)){ close(fd); remove(TMPF); return 1; }
|
||||
|
||||
for(int q=0;q<NE;q++){ compat_aligned_free(m.ws[q].slab); free(m.ws[q].fslab); }
|
||||
for(int i=0;i<m.S.n;i++) free(m.S.t[i].name);
|
||||
free(m.S.t);
|
||||
close(fd);
|
||||
remove(TMPF);
|
||||
puts("test_pipe_block: ok");
|
||||
return 0;
|
||||
}
|
||||
@@ -119,7 +119,7 @@ int main(void){
|
||||
for(int i=0;i<O;i++) sc[i]=0.01f+0.001f*(i%7);
|
||||
for(size_t i=0;i<(size_t)S*K;i++) x[i]=rndf();
|
||||
ColiCudaTensor *t=NULL;
|
||||
ok&=coli_cuda_matmul(&t,ref,x,w4,sc,2,S,K,O,0); /* host path = reference */
|
||||
ok&=coli_cuda_matmul(&t,ref,x,w4,sc,2,S,K,O,0,0); /* host path = reference */
|
||||
float *xd=(float*)coli_cuda_pipe_alloc(0,(size_t)S*K*4);
|
||||
float *yd=(float*)coli_cuda_pipe_alloc(0,(size_t)S*O*4);
|
||||
ok&=coli_cuda_pipe_upload(0,xd,x,(size_t)S*K*4);
|
||||
|
||||
@@ -5,6 +5,7 @@ import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
from resource_plan import (
|
||||
GB,
|
||||
@@ -14,6 +15,7 @@ from resource_plan import (
|
||||
environment_for_plan,
|
||||
format_plan,
|
||||
memory_available,
|
||||
physical_cpu_count,
|
||||
)
|
||||
|
||||
|
||||
@@ -87,6 +89,79 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertIn("clamped", plan["warnings"][0])
|
||||
self.assertIn("0:test-gpu", format_plan(plan))
|
||||
|
||||
def test_auto_tier_thread_count_uses_physical_cores(self):
|
||||
# End-to-end for #325: build_plan + environment_for_plan must export the
|
||||
# physical (not logical SMT) core count as OMP_NUM_THREADS. The original
|
||||
# suite passed physical_cpus=24 explicitly, so it never exercised the
|
||||
# real physical_cpu_count() probe whose single-core failure pinned decode.
|
||||
def lscpu(stdout):
|
||||
return subprocess.CompletedProcess(args=[], returncode=0,
|
||||
stdout=stdout, stderr="")
|
||||
# 1 socket, 12 cores, 2 SMT siblings -> 24 threads, 12 physical cores.
|
||||
|
||||
# The parser must return 12 physical cores under BOTH lscpu layouts:
|
||||
# - 2-col: `lscpu -p=core,socket` emits exactly [core,socket] (this is
|
||||
# what the probe actually requests; the previous fields[1]/[2]
|
||||
# indexing skipped every line here and fell through to the
|
||||
# logical count -> the regression JustVugg caught).
|
||||
# - 3-col: bare `lscpu -p` prepends a CPU column -> [cpu,core,socket].
|
||||
# Taking the last two fields is correct in both cases.
|
||||
layouts = {
|
||||
"2-col (-p=core,socket)": (
|
||||
"# core,socket\n" +
|
||||
"\n".join(f"{core},0" for core in range(12) for _ in range(2))),
|
||||
"3-col (bare -p, CPU prefix)": (
|
||||
"# CPU,Core,Socket\n" +
|
||||
"\n".join(f"{cpu},{core},0" for core in range(12) for cpu in range(2))),
|
||||
}
|
||||
for label, blob in layouts.items():
|
||||
with mock.patch("resource_plan.subprocess.run",
|
||||
return_value=lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
plan = build_plan(self.model, available_memory=16 * GB,
|
||||
available_disk=1, gpus=[])
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(plan["cpu"]["physical_cores"], 12, label)
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], "12", label)
|
||||
|
||||
def test_plan_does_not_set_omp_affinity_vars(self):
|
||||
# The real #325 regression: --auto-tier set OMP_PROC_BIND=spread +
|
||||
# OMP_PLACES=cores, which ran before the engine's overwrite=0 setenv and
|
||||
# so won, collapsing the OpenMP team to one CPU on the reporter's 64-core
|
||||
# Linux box even though OMP_NUM_THREADS was correct. The plan must leave
|
||||
# affinity to the engine's own hot-thread tuning (which prefers 'close').
|
||||
plan = build_plan(self.model, available_memory=16 * GB,
|
||||
available_disk=1, gpus=[], physical_cpus=64)
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], "64")
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
|
||||
def test_plan_conserves_budget_and_experts_above_256gb(self):
|
||||
# Regression for #325's reporter: a 512 GB machine loading the whole
|
||||
# model into RAM. Verify the budget math stays exact at large RAM sizes
|
||||
# (no integer truncation, no over-allocation, no experts lost between
|
||||
# tiers). Checked at 256/512/800 GB to bracket the reporter's box.
|
||||
for ram_gb in (256, 512, 800):
|
||||
plan = build_plan(self.model, ram_gb=ram_gb, available_disk=1,
|
||||
gpus=[], physical_cpus=64)
|
||||
ram = plan["tiers"]["ram"]
|
||||
# RAM budget never over-allocated: dense + runtime + cache <= budget.
|
||||
allocated = (ram["dense_bytes"] + ram["runtime_bytes"]
|
||||
+ ram["expert_cache_bytes"])
|
||||
self.assertLessEqual(allocated, ram["budget_bytes"],
|
||||
f"over-allocated RAM at {ram_gb} GB")
|
||||
# Every expert byte is accounted for exactly once across the tiers.
|
||||
tiers = plan["tiers"]
|
||||
tiered = (tiers["vram"]["hot_expert_bytes"]
|
||||
+ ram["warm_expert_bytes"]
|
||||
+ tiers["disk"]["cold_expert_bytes"])
|
||||
self.assertEqual(tiered, plan["model"]["expert_bytes"],
|
||||
f"expert bytes lost/duplicated at {ram_gb} GB")
|
||||
# A positive RAM budget yields a non-negative cache and a sensible cap.
|
||||
self.assertGreaterEqual(ram["expert_cache_bytes"], 0)
|
||||
self.assertGreaterEqual(ram["cache_slots_per_layer"], 0)
|
||||
|
||||
def test_filters_requested_devices(self):
|
||||
gpus = [{"index": 0, "name": "a", "total_bytes": 8 * GB, "free_bytes": 8 * GB}]
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
@@ -117,13 +192,13 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertEqual(env["COLI_CUDA"], "1")
|
||||
self.assertEqual(env["COLI_GPUS"], "1")
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], str(plan["cpu"]["physical_cores"]))
|
||||
if sys.platform == "win32":
|
||||
# MinGW libgomp: niente affinity su Windows, le chiavi non vanno emesse
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
else:
|
||||
self.assertEqual(env["OMP_PROC_BIND"], "spread")
|
||||
self.assertEqual(env["OMP_PLACES"], "cores")
|
||||
# The plan must NOT set OMP_PROC_BIND / OMP_PLACES on any platform:
|
||||
# the engine's own hot-thread tuning owns affinity (it prefers
|
||||
# OMP_PROC_BIND=close for the back-to-back per-expert matmuls). Setting
|
||||
# spread + cores here ran before the engine's overwrite=0 setenv and so
|
||||
# won, collapsing the team to one CPU on some libgomp topologies (#325).
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
self.assertEqual(env["PIN_GB"], env["CUDA_EXPERT_GB"])
|
||||
|
||||
explicit_threads = environment_for_plan(plan, {"OMP_NUM_THREADS": "7",
|
||||
@@ -140,6 +215,81 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
gpus=[], physical_cpus=8, cpu_sockets=1)
|
||||
self.assertNotIn("COLI_NUMA", environment_for_plan(plan))
|
||||
|
||||
def test_auto_tune_mtp_off_when_compute_bound(self):
|
||||
# Tiny model with 64 GB RAM and no GPU: all experts fit in RAM with no
|
||||
# warm tier, so the plan classifies as compute-bound.
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=24,
|
||||
cpu_sockets=2)
|
||||
# With such a small model fully in RAM and no GPU, bottleneck is compute
|
||||
self.assertEqual(plan["bottleneck_class"], "compute")
|
||||
self.assertIn("DRAFT", plan["tune"])
|
||||
self.assertEqual(plan["tune"]["DRAFT"]["value"], "0")
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["DRAFT"], "0")
|
||||
explicit = environment_for_plan(plan, {"DRAFT": "3"})
|
||||
self.assertEqual(explicit["DRAFT"], "3")
|
||||
|
||||
def test_auto_tune_mtp_off_when_disk_low_hit(self):
|
||||
# Use a model large enough that 8 GB RAM can't hold all experts.
|
||||
big = tempfile.TemporaryDirectory()
|
||||
bigmodel = Path(big.name)
|
||||
(bigmodel / "config.json").write_text(json.dumps({
|
||||
"num_hidden_layers": 2, "n_routed_experts": 4,
|
||||
"kv_lora_rank": 4, "qk_rope_head_dim": 2,
|
||||
"qk_nope_head_dim": 3, "v_head_dim": 5, "num_attention_heads": 2,
|
||||
}))
|
||||
expert_size = 3 * GB # each expert 3 GB → 12 GB total, won't fit in 8 GB budget
|
||||
write_shard(bigmodel / "out-00000.safetensors", [
|
||||
("model.embed_tokens.weight", 100),
|
||||
("model.layers.0.self_attn.q_a_proj.weight", 200),
|
||||
])
|
||||
for i in range(4):
|
||||
write_shard(bigmodel / f"out-{i+1:05d}.safetensors", [
|
||||
(f"model.layers.1.mlp.experts.{i}.gate_proj.weight", expert_size),
|
||||
])
|
||||
plan = build_plan(bigmodel, ram_gb=0, available_memory=4 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=8,
|
||||
cpu_sockets=1)
|
||||
big.cleanup()
|
||||
self.assertEqual(plan["bottleneck_class"], "disk")
|
||||
self.assertLess(plan["projected_hit_rate"], 0.90)
|
||||
self.assertEqual(plan["tune"]["DRAFT"]["value"], "0")
|
||||
|
||||
def test_auto_tune_pipe_multi_gpu(self):
|
||||
gpus = [
|
||||
{"index": 0, "name": "a", "total_bytes": 32 * GB, "free_bytes": 30 * GB},
|
||||
{"index": 1, "name": "b", "total_bytes": 32 * GB, "free_bytes": 30 * GB},
|
||||
]
|
||||
plan = build_plan(self.model, ram_gb=16, available_memory=32 * GB,
|
||||
available_disk=1, gpus=gpus, cpu_sockets=2)
|
||||
self.assertEqual(plan["tune"]["COLI_CUDA_PIPE"]["value"], "2")
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["COLI_CUDA_PIPE"], "2")
|
||||
|
||||
def test_auto_tune_pipe_single_gpu(self):
|
||||
gpus = [{"index": 0, "name": "a", "total_bytes": 12 * GB, "free_bytes": 10 * GB}]
|
||||
plan = build_plan(self.model, ram_gb=16, available_memory=32 * GB,
|
||||
available_disk=1, gpus=gpus, cpu_sockets=1)
|
||||
self.assertEqual(plan["tune"]["COLI_CUDA_PIPE"]["value"], "1")
|
||||
|
||||
def test_auto_tune_numa_hint_for_cpu_only(self):
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=1, gpus=[], physical_cpus=64, cpu_sockets=2)
|
||||
self.assertIn("_numa_hint", plan["tune"])
|
||||
self.assertIn("numactl", plan["tune"]["_numa_hint"])
|
||||
self.assertIn("auto-tune", format_plan(plan))
|
||||
|
||||
def test_format_plan_shows_tune_and_hit_rate(self):
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=24,
|
||||
cpu_sockets=1)
|
||||
text = format_plan(plan)
|
||||
self.assertIn("hit", text)
|
||||
self.assertIn("auto-tune", text)
|
||||
self.assertIn("DRAFT", text)
|
||||
|
||||
def test_cpu_binary_does_not_apply_gpu_tier(self):
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
gpus=[{"index": 0, "name": "a", "total_bytes": 8 * GB,
|
||||
@@ -178,5 +328,73 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertIn("expected_bottleneck", plan)
|
||||
|
||||
|
||||
class PhysicalCpuCountTest(unittest.TestCase):
|
||||
"""Regression for #325: --auto-tier pinned decode to one core because
|
||||
physical_cpu_count() silently returned 1.
|
||||
|
||||
Two root causes this locks down:
|
||||
1. lscpu -p prepends a CPU column, so `-p=core,socket` emits
|
||||
CPU,Core,Socket; counting rows counted logical SMT siblings.
|
||||
2. any probe failure fell through to ``os.cpu_count() or 1`` and the
|
||||
``or 1`` could pin a constrained/cgroup'd box to a single core.
|
||||
"""
|
||||
|
||||
def _lscpu(self, stdout):
|
||||
return subprocess.CompletedProcess(args=[], returncode=0,
|
||||
stdout=stdout, stderr="")
|
||||
|
||||
def _lscpu_topology(self, sockets, cores_per_socket, threads_per_core):
|
||||
# Real lscpu shape: socket-local core IDs repeat across sockets; the
|
||||
# CPU column (always prepended) is a unique logical-CPU index.
|
||||
rows, cpu = [], 0
|
||||
for sock in range(sockets):
|
||||
for core in range(cores_per_socket):
|
||||
for _ in range(threads_per_core):
|
||||
rows.append(f"{cpu},{core},{sock}")
|
||||
cpu += 1
|
||||
return "# CPU,Core,Socket\n" + "\n".join(rows)
|
||||
|
||||
def test_counts_physical_cores_not_smt_threads(self):
|
||||
blob = self._lscpu_topology(sockets=2, cores_per_socket=16, threads_per_core=2)
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 32)
|
||||
|
||||
def test_single_socket_no_smt(self):
|
||||
blob = self._lscpu_topology(sockets=1, cores_per_socket=8, threads_per_core=1)
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 8)
|
||||
|
||||
def test_skips_offline_core_socket_fields(self):
|
||||
# VMs / large NUMA boxes emit "-" for offline core or socket IDs; that
|
||||
# used to raise ValueError, discard the whole parse, and fall through
|
||||
# to the single-core fallback.
|
||||
blob = "# CPU,Core,Socket\n0,0,0\n1,-,0\n2,1,0\n3,1,0\n"
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 2)
|
||||
|
||||
def test_lscpu_missing_falls_back_to_logical_not_silent_one(self):
|
||||
# The bug: lscpu absent -> os.cpu_count() or 1. On a constrained box
|
||||
# os.cpu_count() can be 1. We still must never silently pick 1 without
|
||||
# a warning, and when logical cores exist they must be used.
|
||||
import os
|
||||
with mock.patch("resource_plan.subprocess.run", side_effect=FileNotFoundError), \
|
||||
mock.patch.object(sys, "platform", "linux"), \
|
||||
mock.patch("resource_plan.os.cpu_count", return_value=16), \
|
||||
mock.patch("sys.stderr"):
|
||||
self.assertEqual(physical_cpu_count(), 16)
|
||||
|
||||
def test_zero_logical_cores_warns_and_returns_one(self):
|
||||
# The genuine degenerate case: no probe works and os.cpu_count() is
|
||||
# None/1. Must return 1 (engine needs a positive team size) but warn.
|
||||
with mock.patch("resource_plan.subprocess.run", side_effect=FileNotFoundError), \
|
||||
mock.patch.object(sys, "platform", "linux"), \
|
||||
mock.patch("resource_plan.os.cpu_count", return_value=None), \
|
||||
mock.patch("sys.stderr"):
|
||||
self.assertEqual(physical_cpu_count(), 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
/* Device-router kernel oracle (#431 PR-A).
|
||||
*
|
||||
* Feeds random activations/router weights through pipe_router_logits +
|
||||
* pipe_router_select and checks against a CPU reference that replicates
|
||||
* moe()'s plain routing path verbatim (sigmoid -> bias-augmented top-K by
|
||||
* `choice`, weights from raw `logit`, route-level TOPP truncation, norm_topk,
|
||||
* routed_scale). The dot/expf rounding may differ from libm at ~1e-6 rel, so
|
||||
* a handful of near-tie index flips across trials is tolerated; the weight
|
||||
* math itself must agree to 1e-4 rel on matching selections.
|
||||
*
|
||||
* Build: nvcc -O2 -std=c++17 -arch=native tests/test_router_cuda.cu -o tests/test_router_cuda
|
||||
*/
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <cmath>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
/* pull in the kernel definitions (same idiom as the CPU tests' #include "../colibri.c") */
|
||||
#include "../backend_cuda.cu"
|
||||
|
||||
static void cpu_ref(const float *x,const float *W,const float *bias,int D,int E,
|
||||
int Ksel,float topp,int norm_topk,float rscale,
|
||||
int *idx,float *w,int *keff){
|
||||
float *logit=(float*)malloc(E*sizeof(float)),*choice=(float*)malloc(E*sizeof(float));
|
||||
for(int e=0;e<E;e++){ double a=0; const float *r=W+(size_t)e*D;
|
||||
for(int i=0;i<D;i++) a+=(double)x[i]*r[i];
|
||||
float lg=1.f/(1.f+expf(-(float)a)); logit[e]=lg; choice[e]=lg+bias[e]; }
|
||||
for(int kk=0;kk<Ksel;kk++){ int best=-1; float bv=-1e30f;
|
||||
for(int e=0;e<E;e++){ int tk=0; for(int j=0;j<kk;j++) if(idx[j]==e){tk=1;break;}
|
||||
if(!tk && choice[e]>bv){bv=choice[e];best=e;} }
|
||||
idx[kk]=best; w[kk]=logit[best]; }
|
||||
int Ke=Ksel;
|
||||
if(topp>0.f && topp<1.f){
|
||||
for(int a=1;a<Ksel;a++){ int ii=idx[a]; float ww=w[a]; int b=a-1;
|
||||
while(b>=0 && w[b]<ww){ w[b+1]=w[b]; idx[b+1]=idx[b]; b--; } w[b+1]=ww; idx[b+1]=ii; }
|
||||
float tot=1e-20f; for(int kk=0;kk<Ksel;kk++) tot+=w[kk];
|
||||
float cum=0; for(int kk=0;kk<Ksel;kk++){ cum+=w[kk]; if(cum>=topp*tot){ Ke=kk+1; break; } } }
|
||||
if(norm_topk){ float sm=0; for(int kk=0;kk<Ke;kk++) sm+=w[kk]; sm+=1e-20f;
|
||||
for(int kk=0;kk<Ke;kk++) w[kk]/=sm; }
|
||||
for(int kk=0;kk<Ke;kk++) w[kk]*=rscale;
|
||||
*keff=Ke; free(logit); free(choice);
|
||||
}
|
||||
|
||||
int main(void){
|
||||
const int D=6144,E=256,K=8,TRIALS=200;
|
||||
srand(42);
|
||||
float *x,*W,*b; cudaMallocManaged(&x,D*4); cudaMallocManaged(&W,(size_t)E*D*4);
|
||||
cudaMallocManaged(&b,E*4);
|
||||
float *lg,*ch; char *out;
|
||||
cudaMalloc(&lg,E*4); cudaMalloc(&ch,E*4); cudaMalloc(&out,K*8+4);
|
||||
int flips=0, bad=0;
|
||||
for(int t=0;t<TRIALS;t++){
|
||||
float topp = (t%3==1)?0.7f:0.f;
|
||||
int nt = (t%2);
|
||||
float rs = 1.0f+(t%5)*0.25f;
|
||||
for(int i=0;i<D;i++) x[i]=(rand()/(float)RAND_MAX-.5f)*2.f;
|
||||
for(size_t i=0;i<(size_t)E*D;i++) W[i]=(rand()/(float)RAND_MAX-.5f)*.06f;
|
||||
for(int e=0;e<E;e++) b[e]=(rand()/(float)RAND_MAX-.5f)*.02f;
|
||||
pipe_router_logits<<<E,128>>>(x,W,b,D,lg,ch);
|
||||
pipe_router_select<<<1,1>>>(lg,ch,E,K,topp,nt,rs,out);
|
||||
char pack[K*8+4];
|
||||
if(cudaMemcpy(pack,out,sizeof(pack),cudaMemcpyDeviceToHost)!=cudaSuccess){
|
||||
printf("FAIL cuda\n"); return 1; }
|
||||
int gidx[K],gkeff; float gw[K];
|
||||
memcpy(gidx,pack,K*4); memcpy(gw,pack+K*4,K*4); memcpy(&gkeff,pack+K*8,4);
|
||||
int ridx[K],rkeff; float rw[K];
|
||||
cpu_ref(x,W,b,D,E,K,topp,nt,rs,ridx,rw,&rkeff);
|
||||
int mism=0; for(int k2=0;k2<K;k2++) if(gidx[k2]!=ridx[k2]) mism++;
|
||||
if(mism||gkeff!=rkeff){ flips++; continue; } /* near-tie flip: counted, tolerated */
|
||||
for(int k2=0;k2<gkeff;k2++){
|
||||
float ref=rw[k2], d=fabsf(gw[k2]-ref);
|
||||
if(d>1e-4f*(fabsf(ref)+1e-6f)+1e-6f){ bad++; break; }
|
||||
}
|
||||
}
|
||||
printf("router oracle: %d trials, %d near-tie flips, %d weight mismatches\n",TRIALS,flips,bad);
|
||||
if(flips>4||bad){ printf("FAIL\n"); return 1; }
|
||||
printf("OK\n"); return 0;
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
* iniettato il token scelto e' l'argmax dei FINITI (mai 0 per default), su ogni posizione
|
||||
* del NaN inclusa lo[0]; (c) nessun NaN sopravvive in g_pbuf. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
#include <stdio.h>
|
||||
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
/* Dual-SSD mirror (st_mirror_init & friends): a second read-only model copy is
|
||||
* accepted only when byte-identical in size and safetensors header, reads on
|
||||
* either replica return the same bytes, and divergent/missing copies degrade
|
||||
* to the primary instead of being trusted. Fixture dirs are created in the
|
||||
* current working directory and removed on exit. */
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "../st.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <direct.h>
|
||||
#define MKDIR(p) _mkdir(p)
|
||||
#else
|
||||
#include <sys/stat.h>
|
||||
#define MKDIR(p) mkdir(p, 0777)
|
||||
#endif
|
||||
|
||||
#define CHECK(condition) do { \
|
||||
if (!(condition)) { \
|
||||
fprintf(stderr, "%s:%d: check failed: %s\n", __FILE__, __LINE__, #condition); \
|
||||
return 1; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define DIR_A "tmp_mirror_a" /* primary */
|
||||
#define DIR_B "tmp_mirror_b" /* identical copy */
|
||||
#define DIR_C "tmp_mirror_c" /* same size, divergent header */
|
||||
#define DIR_D "tmp_mirror_d" /* different size */
|
||||
#define DIR_E "tmp_mirror_e" /* empty (missing file) */
|
||||
|
||||
/* one-tensor safetensors file; flip lets us corrupt one header byte and
|
||||
* pad lets us grow the payload, both without changing anything else */
|
||||
static int write_model(const char *dir, int flip, int pad) {
|
||||
char hdr[128], path[256];
|
||||
int hl = snprintf(hdr, sizeof(hdr),
|
||||
"{\"t0\":{\"dtype\":\"F32\",\"shape\":[8],\"data_offsets\":[0,32]}}");
|
||||
if (flip) hdr[hl - 2] = ' '; /* inside the JSON header, same length */
|
||||
snprintf(path, sizeof(path), "%s/model.safetensors", dir);
|
||||
FILE *f = fopen(path, "wb");
|
||||
if (!f) { perror(path); return -1; }
|
||||
uint64_t n = (uint64_t)hl;
|
||||
fwrite(&n, 8, 1, f);
|
||||
fwrite(hdr, 1, (size_t)hl, f);
|
||||
float data[8] = {1, -2, 3.5f, 0, 42, -0.5f, 7, 8};
|
||||
fwrite(data, 4, 8, f);
|
||||
for (int i = 0; i < pad; i++) fputc(0, f);
|
||||
fclose(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void cleanup(void) {
|
||||
const char *dirs[] = {DIR_A, DIR_B, DIR_C, DIR_D, DIR_E};
|
||||
for (int i = 0; i < 5; i++) {
|
||||
char path[256];
|
||||
snprintf(path, sizeof(path), "%s/model.safetensors", dirs[i]);
|
||||
remove(path);
|
||||
remove(dirs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
cleanup();
|
||||
CHECK(MKDIR(DIR_A) == 0 && MKDIR(DIR_B) == 0 && MKDIR(DIR_C) == 0 &&
|
||||
MKDIR(DIR_D) == 0 && MKDIR(DIR_E) == 0);
|
||||
CHECK(write_model(DIR_A, 0, 0) == 0);
|
||||
CHECK(write_model(DIR_B, 0, 0) == 0);
|
||||
CHECK(write_model(DIR_C, 1, 0) == 0);
|
||||
CHECK(write_model(DIR_D, 0, 64) == 0);
|
||||
|
||||
shards S;
|
||||
st_init(&S, DIR_A);
|
||||
CHECK(S.n == 1 && S.nfd == 1);
|
||||
st_tensor *t = st_find(&S, "t0");
|
||||
CHECK(t != NULL);
|
||||
|
||||
/* without a mirror: replica 0 is the identity, replica 1 is absent */
|
||||
CHECK(st_fd_rep(&S, t->fd, 0) == t->fd);
|
||||
CHECK(st_fd_rep(&S, t->fd, 1) == -1);
|
||||
|
||||
/* identical copy: accepted, and both replicas serve the same bytes */
|
||||
CHECK(st_mirror_init(&S, DIR_B) == 1);
|
||||
int mfd = st_fd_rep(&S, t->fd, 1);
|
||||
CHECK(mfd >= 0 && mfd != t->fd);
|
||||
float a[8], b[8];
|
||||
CHECK(pread(t->fd, a, t->nbytes, t->off) == t->nbytes);
|
||||
CHECK(pread(mfd, b, t->nbytes, t->off) == t->nbytes);
|
||||
CHECK(memcmp(a, b, sizeof(a)) == 0);
|
||||
st_prefetch_rep(&S, "t0", 1); /* smoke: WILLNEED on the mirror fd */
|
||||
st_prefetch_rep(&S, "t0", 0);
|
||||
|
||||
/* divergent header, same size: rejected */
|
||||
CHECK(st_mirror_init(&S, DIR_C) == 0);
|
||||
CHECK(st_fd_rep(&S, t->fd, 1) == -1);
|
||||
|
||||
/* different size: rejected */
|
||||
CHECK(st_mirror_init(&S, DIR_D) == 0);
|
||||
|
||||
/* missing file: rejected (partial mirror with zero shards) */
|
||||
CHECK(st_mirror_init(&S, DIR_E) == 0);
|
||||
|
||||
/* unknown fd never maps to a replica */
|
||||
CHECK(st_fd_rep(&S, 987654, 1) == -1);
|
||||
|
||||
cleanup();
|
||||
puts("safetensors mirror tests: ok");
|
||||
return 0;
|
||||
}
|
||||
@@ -20,7 +20,7 @@
|
||||
* Defense 2 is what makes this robust against checkpoints we don't control:
|
||||
* even with BOTH configs mutilated, a control token cannot leak into a reply. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static const char *TOKJSON =
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
/* o200k pre-tokenizer validation against HF-tokenizers-generated expectations.
|
||||
* Self-contained for the test-c harness: loads tests/tok_o200k_tiny.json (a
|
||||
* synthetic byte-level BPE whose Split regex is the o200k pattern — a few KB,
|
||||
* no model download) and scores tests/tok_o200k_cases.txt, whose expected ids
|
||||
* were produced by HF `tokenizers` on the same file. Guards the case-aware
|
||||
* letter matcher, contractions, digit groups, the [\r\n/]* punctuation tail,
|
||||
* whitespace branches, and added-token atomicity; round-trips every case.
|
||||
* The cl100k path is untouched by construction (dispatch requires \p{Lu} in
|
||||
* the tokenizer's own Split pattern) and stays covered by the GLM oracle. */
|
||||
#define _GNU_SOURCE
|
||||
#include "../tok.h"
|
||||
|
||||
int main(void) {
|
||||
Tok T;
|
||||
tok_load(&T, "tests/tok_o200k_tiny.json");
|
||||
if (!T.o200k) { fprintf(stderr, "test_tok_o200k: o200k pattern not detected\n"); return 1; }
|
||||
FILE *f = fopen("tests/tok_o200k_cases.txt", "rb");
|
||||
if (!f) { perror("tests/tok_o200k_cases.txt"); return 1; }
|
||||
/* fgets, not getline: MinGW's UCRT lacks getline and this must run on
|
||||
* the windows job. Case lines are short; 8 KB is generous. */
|
||||
char line[8192];
|
||||
int pass = 0, tot = 0, dpass = 0;
|
||||
while (fgets(line, sizeof(line), f)) {
|
||||
size_t nr = strlen(line);
|
||||
while (nr > 0 && (line[nr-1] == '\n' || line[nr-1] == '\r')) line[--nr] = 0;
|
||||
if (nr == 0) continue;
|
||||
char *tab = strchr(line, '\t'); if (!tab) continue;
|
||||
*tab = 0;
|
||||
const char *text = line, *idstr = tab + 1;
|
||||
char tbuf[4096]; int tn = 0;
|
||||
for (const char *q = text; *q && tn < 4095; q++) {
|
||||
if (q[0]=='\\' && q[1]=='n') { tbuf[tn++]='\n'; q++; }
|
||||
else if (q[0]=='\\' && q[1]=='t') { tbuf[tn++]='\t'; q++; }
|
||||
else if (q[0]=='\\' && q[1]=='r') { tbuf[tn++]='\r'; q++; }
|
||||
else if (q[0]=='\\' && q[1]=='\\') { tbuf[tn++]='\\'; q++; }
|
||||
else tbuf[tn++] = *q;
|
||||
}
|
||||
tbuf[tn] = 0;
|
||||
int exp[512], ne = 0;
|
||||
for (const char *q = idstr; *q; ) {
|
||||
while (*q == ',' || *q == ' ') q++;
|
||||
if (!*q) break;
|
||||
exp[ne++] = atoi(q);
|
||||
while (*q && *q != ',') q++;
|
||||
}
|
||||
int got[512]; int ng = tok_encode(&T, tbuf, tn, got, 512);
|
||||
int ok = (ng == ne);
|
||||
for (int i = 0; i < ng && ok; i++) ok = (got[i] == exp[i]);
|
||||
tot++; if (ok) pass++;
|
||||
char dec[8192]; int dn = tok_decode(&T, got, ng, dec, 8191);
|
||||
int drt = (dn == tn) && !memcmp(dec, tbuf, tn);
|
||||
if (drt) dpass++;
|
||||
if (!ok || !drt) {
|
||||
fprintf(stderr, "MISMATCH text=%s\n exp(%d):", text, ne);
|
||||
for (int i = 0; i < ne; i++) fprintf(stderr, " %d", exp[i]);
|
||||
fprintf(stderr, "\n got(%d):", ng);
|
||||
for (int i = 0; i < ng; i++) fprintf(stderr, " %d", got[i]);
|
||||
fprintf(stderr, "\n decode_ok=%d\n", drt);
|
||||
}
|
||||
}
|
||||
fclose(f);
|
||||
printf("test_tok_o200k: ENCODE %d/%d DECODE %d/%d\n", pass, tot, dpass, tot);
|
||||
return (pass == tot && dpass == tot) ? 0 : 2;
|
||||
}
|
||||
+1
-1
@@ -24,7 +24,7 @@
|
||||
* No scratch files: the test runs entirely in memory (no mkdtemp), so it builds clean on
|
||||
* the Windows MinGW CI job without the unmerged compat shim (#352). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static int fail(const char *s){ fprintf(stderr,"FAIL: %s\n",s); return 1; }
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
hello world 259,32,119,111,114,263
|
||||
HelloWorld 72,101,257,111,262
|
||||
XMLHttpRequest 88,77,76,72,116,116,112,82,101,113,117,101,115,116
|
||||
helloWORLDhello 259,87,79,82,76,68,259
|
||||
dog's 100,111,103,270
|
||||
DON'T 68,79,78,39,84
|
||||
don't 100,111,110,39,116
|
||||
I'll've 73,39,257,39,118,101
|
||||
O'Brien 79,39,66,114,105,101,110
|
||||
the theatre 265,32,116,256,97,116,114,101
|
||||
12345 268,52,53
|
||||
a1b22c333d4444 97,49,98,50,50,99,51,51,51,100,52,52,52,52
|
||||
3.14 51,46,49,52
|
||||
http://x.com/a/b 104,116,116,112,58,47,47,120,46,99,111,109,47,97,47,98
|
||||
path/to/file 112,97,264,47,116,111,47,102,105,108,101
|
||||
a//b///c 97,47,47,98,47,47,47,99
|
||||
!!\r\n//x 33,33,13,10,47,47,120
|
||||
one\ntwo\r\nthree 111,110,101,10,116,119,111,13,10,264,114,101,101
|
||||
\n x 32,32,10,32,32,120
|
||||
32,32,32
|
||||
a b c 97,32,32,98,32,32,32,99
|
||||
tab\there 116,269,9,256,114,101
|
||||
Café 67,97,102,101,204,129
|
||||
naiveBayes 110,97,105,118,101,66,97,121,101,115
|
||||
Éclair 195,137,99,108,97,105,114
|
||||
北京大学 229,140,151,228,186,172,229,164,167,229,173,166
|
||||
ΑΒαβ 206,145,206,146,206,177,206,178
|
||||
Иван 208,152,208,178,208,176,208,189
|
||||
ẞßscharf 225,186,158,195,159,115,99,104,97,114,102
|
||||
i̇stanbul 105,204,135,115,116,97,110,98,117,108
|
||||
hello<|endoftext|>world 259,274,119,111,114,263
|
||||
<|message_user|>hi 275,104,105
|
||||
mixedCASEand123 109,105,120,101,100,67,65,83,69,97,110,100,268
|
||||
's 270
|
||||
's 32,39,115
|
||||
A 65
|
||||
aB 97,66
|
||||
Ab 65,98
|
||||
AB 65,66
|
||||
ab 269
|
||||
@@ -0,0 +1 @@
|
||||
{"version": "1.0", "truncation": null, "padding": null, "added_tokens": [{"id": 274, "content": "<|endoftext|>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true}, {"id": 275, "content": "<|message_user|>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true}], "normalizer": null, "pre_tokenizer": {"type": "Sequence", "pretokenizers": [{"type": "Split", "pattern": {"Regex": "[^\\r\\n\\p{L}\\p{N}]?[\\p{Lu}\\p{Lt}\\p{Lm}\\p{Lo}\\p{M}]*[\\p{Ll}\\p{Lm}\\p{Lo}\\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?|[^\\r\\n\\p{L}\\p{N}]?[\\p{Lu}\\p{Lt}\\p{Lm}\\p{Lo}\\p{M}]+[\\p{Ll}\\p{Lm}\\p{Lo}\\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"}, "behavior": "Isolated", "invert": false}, {"type": "ByteLevel", "add_prefix_space": false, "trim_offsets": true, "use_regex": false}]}, "post_processor": null, "decoder": {"type": "ByteLevel", "add_prefix_space": true, "trim_offsets": true, "use_regex": true}, "model": {"type": "BPE", "dropout": null, "unk_token": null, "continuing_subword_prefix": null, "end_of_word_suffix": null, "fuse_unk": false, "byte_fallback": false, "ignore_merges": true, "vocab": {"Ā": 0, "ā": 1, "Ă": 2, "ă": 3, "Ą": 4, "ą": 5, "Ć": 6, "ć": 7, "Ĉ": 8, "ĉ": 9, "Ċ": 10, "ċ": 11, "Č": 12, "č": 13, "Ď": 14, "ď": 15, "Đ": 16, "đ": 17, "Ē": 18, "ē": 19, "Ĕ": 20, "ĕ": 21, "Ė": 22, "ė": 23, "Ę": 24, "ę": 25, "Ě": 26, "ě": 27, "Ĝ": 28, "ĝ": 29, "Ğ": 30, "ğ": 31, "Ġ": 32, "!": 33, "\"": 34, "#": 35, "$": 36, "%": 37, "&": 38, "'": 39, "(": 40, ")": 41, "*": 42, "+": 43, ",": 44, "-": 45, ".": 46, "/": 47, "0": 48, "1": 49, "2": 50, "3": 51, "4": 52, "5": 53, "6": 54, "7": 55, "8": 56, "9": 57, ":": 58, ";": 59, "<": 60, "=": 61, ">": 62, "?": 63, "@": 64, "A": 65, "B": 66, "C": 67, "D": 68, "E": 69, "F": 70, "G": 71, "H": 72, "I": 73, "J": 74, "K": 75, "L": 76, "M": 77, "N": 78, "O": 79, "P": 80, "Q": 81, "R": 82, "S": 83, "T": 84, "U": 85, "V": 86, "W": 87, "X": 88, "Y": 89, "Z": 90, "[": 91, "\\": 92, "]": 93, "^": 94, "_": 95, "`": 96, "a": 97, "b": 98, "c": 99, "d": 100, "e": 101, "f": 102, "g": 103, "h": 104, "i": 105, "j": 106, "k": 107, "l": 108, "m": 109, "n": 110, "o": 111, "p": 112, "q": 113, "r": 114, "s": 115, "t": 116, "u": 117, "v": 118, "w": 119, "x": 120, "y": 121, "z": 122, "{": 123, "|": 124, "}": 125, "~": 126, "ġ": 127, "Ģ": 128, "ģ": 129, "Ĥ": 130, "ĥ": 131, "Ħ": 132, "ħ": 133, "Ĩ": 134, "ĩ": 135, "Ī": 136, "ī": 137, "Ĭ": 138, "ĭ": 139, "Į": 140, "į": 141, "İ": 142, "ı": 143, "IJ": 144, "ij": 145, "Ĵ": 146, "ĵ": 147, "Ķ": 148, "ķ": 149, "ĸ": 150, "Ĺ": 151, "ĺ": 152, "Ļ": 153, "ļ": 154, "Ľ": 155, "ľ": 156, "Ŀ": 157, "ŀ": 158, "Ł": 159, "ł": 160, "¡": 161, "¢": 162, "£": 163, "¤": 164, "¥": 165, "¦": 166, "§": 167, "¨": 168, "©": 169, "ª": 170, "«": 171, "¬": 172, "Ń": 173, "®": 174, "¯": 175, "°": 176, "±": 177, "²": 178, "³": 179, "´": 180, "µ": 181, "¶": 182, "·": 183, "¸": 184, "¹": 185, "º": 186, "»": 187, "¼": 188, "½": 189, "¾": 190, "¿": 191, "À": 192, "Á": 193, "Â": 194, "Ã": 195, "Ä": 196, "Å": 197, "Æ": 198, "Ç": 199, "È": 200, "É": 201, "Ê": 202, "Ë": 203, "Ì": 204, "Í": 205, "Î": 206, "Ï": 207, "Ð": 208, "Ñ": 209, "Ò": 210, "Ó": 211, "Ô": 212, "Õ": 213, "Ö": 214, "×": 215, "Ø": 216, "Ù": 217, "Ú": 218, "Û": 219, "Ü": 220, "Ý": 221, "Þ": 222, "ß": 223, "à": 224, "á": 225, "â": 226, "ã": 227, "ä": 228, "å": 229, "æ": 230, "ç": 231, "è": 232, "é": 233, "ê": 234, "ë": 235, "ì": 236, "í": 237, "î": 238, "ï": 239, "ð": 240, "ñ": 241, "ò": 242, "ó": 243, "ô": 244, "õ": 245, "ö": 246, "÷": 247, "ø": 248, "ù": 249, "ú": 250, "û": 251, "ü": 252, "ý": 253, "þ": 254, "ÿ": 255, "he": 256, "ll": 257, "hell": 258, "hello": 259, "Wo": 260, "Wor": 261, "World": 262, "ld": 263, "th": 264, "the": 265, "Ġthe": 266, "12": 267, "123": 268, "ab": 269, "'s": 270, "Ġa": 271, "./": 272, "ĊĊ": 273}, "merges": [["h", "e"], ["l", "l"], ["he", "ll"], ["hell", "o"], ["W", "o"], ["Wo", "r"], ["Wor", "ld"], ["l", "d"], ["t", "h"], ["th", "e"], ["Ġ", "the"], ["1", "2"], ["12", "3"], ["a", "b"], ["'", "s"], ["Ġ", "a"], [".", "/"], ["Ċ", "Ċ"]]}}
|
||||
@@ -19,6 +19,7 @@
|
||||
#include <limits.h>
|
||||
#include "json.h"
|
||||
#include "tok_unicode.h"
|
||||
#include "tok_unicode_o200k.h"
|
||||
|
||||
/* ---------- hash map (chiavi binarie con lunghezza) ---------- */
|
||||
typedef struct { const char *k; int klen; int v; int used; } ment;
|
||||
@@ -50,6 +51,7 @@ typedef struct {
|
||||
Special *sp; int nsp; /* added tokens, ordinati per lunghezza decrescente */
|
||||
uint32_t byte2cp[256]; int byte2cp_len[256]; char byte2str[256][3];
|
||||
int16_t cp2byte[1024];
|
||||
int o200k; /* pre_tokenizer regex family: 0 = cl100k (GLM), 1 = o200k (Inkling) */
|
||||
} Tok;
|
||||
|
||||
/* ---------- UTF-8 ---------- */
|
||||
@@ -144,6 +146,17 @@ static void tok_load(Tok *T, const char *path){
|
||||
}
|
||||
qsort(T->sp,T->nsp,sizeof(Special),cmp_sp_len); /* match piu' lungo per primo */
|
||||
}
|
||||
/* pre_tokenizer family: the o200k Split regex is recognizable by its
|
||||
* case-category classes (\p{Lu}...) which cl100k does not use */
|
||||
jval *pt=json_get(root,"pre_tokenizer");
|
||||
if(pt){
|
||||
jval *ps=json_get(pt,"pretokenizers");
|
||||
if(ps&&ps->t==J_ARR) for(int i=0;i<ps->len;i++){
|
||||
jval *pat=json_get(ps->kids[i],"pattern");
|
||||
jval *rx=pat?json_get(pat,"Regex"):NULL;
|
||||
if(rx&&rx->t==J_STR&&strstr(rx->str,"\\p{Lu}")) T->o200k=1;
|
||||
}
|
||||
}
|
||||
/* arena/buf restano allocati: le stringhe (j_dup) sono malloc indipendenti e ci servono vive */
|
||||
(void)arena;
|
||||
}
|
||||
@@ -241,6 +254,104 @@ static void pretok_chunk(Tok *T, const unsigned char *p, int a, int b, int *out,
|
||||
free(cp); free(off);
|
||||
}
|
||||
|
||||
/* ---------- pre-tokenizer o200k (Inkling / GPT-4o family) ----------
|
||||
* Split regex:
|
||||
* A: [^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?
|
||||
* B: [^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?
|
||||
* C: \p{N}{1,3} D: ' ?[^\s\p{L}\p{N}]+[\r\n/]*' E: \s*[\r\n]+ F: \s+(?!\S) G: \s+
|
||||
* S1 = Lu|Lt|Lm|Lo|M, S2 = Ll|Lm|Lo|M. The letter matcher below replays the
|
||||
* regex engine's backtracking order exactly: A with greedy optional prefix and
|
||||
* maximally-greedy S1* given back until S2+ can take >=1 char, then B. */
|
||||
#define O2_S1(c) (is_U(c)||is_X(c))
|
||||
#define O2_S2(c) (is_X(c)||(is_L(c)&&!is_U(c)))
|
||||
static uint32_t o2_low(uint32_t c){ return (c>='A'&&c<='Z')?c+32:c; }
|
||||
static int o2_contraction(const uint32_t *cp, int n, int k){
|
||||
if(k<n && cp[k]=='\'' && k+1<n){
|
||||
uint32_t d=o2_low(cp[k+1]);
|
||||
if(k+2<n){ uint32_t e=o2_low(cp[k+2]);
|
||||
if((d=='r'&&e=='e')||(d=='v'&&e=='e')||(d=='l'&&e=='l')) return k+3; }
|
||||
if(d=='s'||d=='t'||d=='m'||d=='d') return k+2;
|
||||
}
|
||||
return k;
|
||||
}
|
||||
/* end (cp index) of branch A|B match at i, or -1 */
|
||||
static int o2_letters(const uint32_t *cp, int n, int i){
|
||||
/* branch A, prefix greedy (taken first), then without prefix */
|
||||
for(int pfx=1; pfx>=0; pfx--){
|
||||
int j0=i;
|
||||
if(pfx){
|
||||
uint32_t c=cp[i];
|
||||
if(c=='\r'||c=='\n'||is_L(c)||is_N(c)||i+1>=n) continue;
|
||||
j0=i+1;
|
||||
}
|
||||
int m1=j0; while(m1<n && O2_S1(cp[m1])) m1++;
|
||||
for(int s=m1; s>=j0; s--){
|
||||
if(s<n && O2_S2(cp[s])){
|
||||
int k=s+1; while(k<n && O2_S2(cp[k])) k++;
|
||||
return o2_contraction(cp,n,k);
|
||||
}
|
||||
}
|
||||
}
|
||||
/* branch B */
|
||||
for(int pfx=1; pfx>=0; pfx--){
|
||||
int j0=i;
|
||||
if(pfx){
|
||||
uint32_t c=cp[i];
|
||||
if(c=='\r'||c=='\n'||is_L(c)||is_N(c)||i+1>=n) continue;
|
||||
j0=i+1;
|
||||
}
|
||||
int m1=j0; while(m1<n && O2_S1(cp[m1])) m1++;
|
||||
if(m1>j0){
|
||||
int k=m1; while(k<n && O2_S2(cp[k])) k++;
|
||||
return o2_contraction(cp,n,k);
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static void pretok_chunk_o200k(Tok *T, const unsigned char *p, int a, int b, int *out, int *no, int max){
|
||||
int nb=b-a; if(nb<=0) return;
|
||||
uint32_t *cp=malloc((nb+1)*sizeof(uint32_t)); int *off=malloc((nb+2)*sizeof(int)); int n=0;
|
||||
for(int i=a;i<b;){ uint32_t c; int k=u8_next(p,b,i,&c); off[n]=i; cp[n]=c; n++; i+=k; }
|
||||
off[n]=b;
|
||||
#define ISNL(c) ((c)=='\r'||(c)=='\n')
|
||||
int i=0;
|
||||
while(i<n){
|
||||
int start=i; uint32_t c=cp[i];
|
||||
/* A|B: letter runs with case-aware split + optional contraction */
|
||||
{
|
||||
int e=o2_letters(cp,n,i);
|
||||
if(e>i){ i=e; bpe_piece(T,p,off[start],off[i],out,no,max); continue; }
|
||||
}
|
||||
/* C: \p{N}{1,3} */
|
||||
if(is_N(c)){ int j=i,k=0; while(j<n && is_N(cp[j]) && k<3){ j++; k++; } i=j; bpe_piece(T,p,off[start],off[i],out,no,max); continue; }
|
||||
/* D: ' ?[^\s\p{L}\p{N}]+[\r\n/]*' */
|
||||
{
|
||||
int j=i;
|
||||
if(c==' ' && j+1<n && !is_S(cp[j+1]) && !is_L(cp[j+1]) && !is_N(cp[j+1])) j++;
|
||||
if(j<n && !is_S(cp[j]) && !is_L(cp[j]) && !is_N(cp[j])){
|
||||
while(j<n && !is_S(cp[j]) && !is_L(cp[j]) && !is_N(cp[j])) j++;
|
||||
while(j<n && (ISNL(cp[j]) || cp[j]=='/')) j++;
|
||||
i=j; bpe_piece(T,p,off[start],off[i],out,no,max); continue;
|
||||
}
|
||||
}
|
||||
/* E: \s*[\r\n]+ F: \s+(?!\S) G: \s+ (same as cl100k) */
|
||||
{
|
||||
int r=i; while(r<n && is_S(cp[r])) r++;
|
||||
if(r>i){ int last=-1; for(int j=i;j<r;j++) if(ISNL(cp[j])) last=j;
|
||||
if(last>=0){ i=last+1; bpe_piece(T,p,off[start],off[i],out,no,max); continue; }
|
||||
int end = (r<n) ? r-1 : r;
|
||||
if(end<=i) end=i+1;
|
||||
i=end; bpe_piece(T,p,off[start],off[i],out,no,max); continue;
|
||||
}
|
||||
}
|
||||
i++;
|
||||
bpe_piece(T,p,off[start],off[i],out,no,max);
|
||||
}
|
||||
#undef ISNL
|
||||
free(cp); free(off);
|
||||
}
|
||||
|
||||
/* ---------- encode: testo -> id (split sugli added token, poi pretok+BPE) ---------- */
|
||||
static int tok_encode(Tok *T, const char *text, int len, int *out, int max){
|
||||
const unsigned char *p=(const unsigned char*)text; int no=0; int i=0;
|
||||
@@ -254,7 +365,10 @@ static int tok_encode(Tok *T, const char *text, int len, int *out, int max){
|
||||
}
|
||||
}
|
||||
int chunk_end = (hitpos<0) ? len : hitpos;
|
||||
if(chunk_end>i) pretok_chunk(T,p,i,chunk_end,out,&no,max);
|
||||
if(chunk_end>i){
|
||||
if(T->o200k) pretok_chunk_o200k(T,p,i,chunk_end,out,&no,max);
|
||||
else pretok_chunk(T,p,i,chunk_end,out,&no,max);
|
||||
}
|
||||
if(hitpos<0) break;
|
||||
if(no<max) out[no++]=hitid;
|
||||
i=hitpos+hitlen;
|
||||
|
||||
@@ -0,0 +1,228 @@
|
||||
/* Unicode range tables for the o200k pre-tokenizer (Inkling, GPT-4o family).
|
||||
* Generated from Python unicodedata (see tools/ in git history):
|
||||
* uni_U = Lu + Lt (uppercase/titlecase letters)
|
||||
* uni_X = Lm + Lo + M (caseless letters and combining marks; members of
|
||||
* BOTH bracket classes in the o200k regex)
|
||||
* Lookup via the same uni_in() binary search as tok_unicode.h. */
|
||||
#ifndef TOK_UNICODE_O200K_H
|
||||
#define TOK_UNICODE_O200K_H
|
||||
#include <stdint.h>
|
||||
|
||||
static const uint32_t uni_U[][2] = {
|
||||
{0x41,0x5A},{0xC0,0xD6},{0xD8,0xDE},{0x100,0x100},{0x102,0x102},{0x104,0x104},
|
||||
{0x106,0x106},{0x108,0x108},{0x10A,0x10A},{0x10C,0x10C},{0x10E,0x10E},{0x110,0x110},
|
||||
{0x112,0x112},{0x114,0x114},{0x116,0x116},{0x118,0x118},{0x11A,0x11A},{0x11C,0x11C},
|
||||
{0x11E,0x11E},{0x120,0x120},{0x122,0x122},{0x124,0x124},{0x126,0x126},{0x128,0x128},
|
||||
{0x12A,0x12A},{0x12C,0x12C},{0x12E,0x12E},{0x130,0x130},{0x132,0x132},{0x134,0x134},
|
||||
{0x136,0x136},{0x139,0x139},{0x13B,0x13B},{0x13D,0x13D},{0x13F,0x13F},{0x141,0x141},
|
||||
{0x143,0x143},{0x145,0x145},{0x147,0x147},{0x14A,0x14A},{0x14C,0x14C},{0x14E,0x14E},
|
||||
{0x150,0x150},{0x152,0x152},{0x154,0x154},{0x156,0x156},{0x158,0x158},{0x15A,0x15A},
|
||||
{0x15C,0x15C},{0x15E,0x15E},{0x160,0x160},{0x162,0x162},{0x164,0x164},{0x166,0x166},
|
||||
{0x168,0x168},{0x16A,0x16A},{0x16C,0x16C},{0x16E,0x16E},{0x170,0x170},{0x172,0x172},
|
||||
{0x174,0x174},{0x176,0x176},{0x178,0x179},{0x17B,0x17B},{0x17D,0x17D},{0x181,0x182},
|
||||
{0x184,0x184},{0x186,0x187},{0x189,0x18B},{0x18E,0x191},{0x193,0x194},{0x196,0x198},
|
||||
{0x19C,0x19D},{0x19F,0x1A0},{0x1A2,0x1A2},{0x1A4,0x1A4},{0x1A6,0x1A7},{0x1A9,0x1A9},
|
||||
{0x1AC,0x1AC},{0x1AE,0x1AF},{0x1B1,0x1B3},{0x1B5,0x1B5},{0x1B7,0x1B8},{0x1BC,0x1BC},
|
||||
{0x1C4,0x1C5},{0x1C7,0x1C8},{0x1CA,0x1CB},{0x1CD,0x1CD},{0x1CF,0x1CF},{0x1D1,0x1D1},
|
||||
{0x1D3,0x1D3},{0x1D5,0x1D5},{0x1D7,0x1D7},{0x1D9,0x1D9},{0x1DB,0x1DB},{0x1DE,0x1DE},
|
||||
{0x1E0,0x1E0},{0x1E2,0x1E2},{0x1E4,0x1E4},{0x1E6,0x1E6},{0x1E8,0x1E8},{0x1EA,0x1EA},
|
||||
{0x1EC,0x1EC},{0x1EE,0x1EE},{0x1F1,0x1F2},{0x1F4,0x1F4},{0x1F6,0x1F8},{0x1FA,0x1FA},
|
||||
{0x1FC,0x1FC},{0x1FE,0x1FE},{0x200,0x200},{0x202,0x202},{0x204,0x204},{0x206,0x206},
|
||||
{0x208,0x208},{0x20A,0x20A},{0x20C,0x20C},{0x20E,0x20E},{0x210,0x210},{0x212,0x212},
|
||||
{0x214,0x214},{0x216,0x216},{0x218,0x218},{0x21A,0x21A},{0x21C,0x21C},{0x21E,0x21E},
|
||||
{0x220,0x220},{0x222,0x222},{0x224,0x224},{0x226,0x226},{0x228,0x228},{0x22A,0x22A},
|
||||
{0x22C,0x22C},{0x22E,0x22E},{0x230,0x230},{0x232,0x232},{0x23A,0x23B},{0x23D,0x23E},
|
||||
{0x241,0x241},{0x243,0x246},{0x248,0x248},{0x24A,0x24A},{0x24C,0x24C},{0x24E,0x24E},
|
||||
{0x370,0x370},{0x372,0x372},{0x376,0x376},{0x37F,0x37F},{0x386,0x386},{0x388,0x38A},
|
||||
{0x38C,0x38C},{0x38E,0x38F},{0x391,0x3A1},{0x3A3,0x3AB},{0x3CF,0x3CF},{0x3D2,0x3D4},
|
||||
{0x3D8,0x3D8},{0x3DA,0x3DA},{0x3DC,0x3DC},{0x3DE,0x3DE},{0x3E0,0x3E0},{0x3E2,0x3E2},
|
||||
{0x3E4,0x3E4},{0x3E6,0x3E6},{0x3E8,0x3E8},{0x3EA,0x3EA},{0x3EC,0x3EC},{0x3EE,0x3EE},
|
||||
{0x3F4,0x3F4},{0x3F7,0x3F7},{0x3F9,0x3FA},{0x3FD,0x42F},{0x460,0x460},{0x462,0x462},
|
||||
{0x464,0x464},{0x466,0x466},{0x468,0x468},{0x46A,0x46A},{0x46C,0x46C},{0x46E,0x46E},
|
||||
{0x470,0x470},{0x472,0x472},{0x474,0x474},{0x476,0x476},{0x478,0x478},{0x47A,0x47A},
|
||||
{0x47C,0x47C},{0x47E,0x47E},{0x480,0x480},{0x48A,0x48A},{0x48C,0x48C},{0x48E,0x48E},
|
||||
{0x490,0x490},{0x492,0x492},{0x494,0x494},{0x496,0x496},{0x498,0x498},{0x49A,0x49A},
|
||||
{0x49C,0x49C},{0x49E,0x49E},{0x4A0,0x4A0},{0x4A2,0x4A2},{0x4A4,0x4A4},{0x4A6,0x4A6},
|
||||
{0x4A8,0x4A8},{0x4AA,0x4AA},{0x4AC,0x4AC},{0x4AE,0x4AE},{0x4B0,0x4B0},{0x4B2,0x4B2},
|
||||
{0x4B4,0x4B4},{0x4B6,0x4B6},{0x4B8,0x4B8},{0x4BA,0x4BA},{0x4BC,0x4BC},{0x4BE,0x4BE},
|
||||
{0x4C0,0x4C1},{0x4C3,0x4C3},{0x4C5,0x4C5},{0x4C7,0x4C7},{0x4C9,0x4C9},{0x4CB,0x4CB},
|
||||
{0x4CD,0x4CD},{0x4D0,0x4D0},{0x4D2,0x4D2},{0x4D4,0x4D4},{0x4D6,0x4D6},{0x4D8,0x4D8},
|
||||
{0x4DA,0x4DA},{0x4DC,0x4DC},{0x4DE,0x4DE},{0x4E0,0x4E0},{0x4E2,0x4E2},{0x4E4,0x4E4},
|
||||
{0x4E6,0x4E6},{0x4E8,0x4E8},{0x4EA,0x4EA},{0x4EC,0x4EC},{0x4EE,0x4EE},{0x4F0,0x4F0},
|
||||
{0x4F2,0x4F2},{0x4F4,0x4F4},{0x4F6,0x4F6},{0x4F8,0x4F8},{0x4FA,0x4FA},{0x4FC,0x4FC},
|
||||
{0x4FE,0x4FE},{0x500,0x500},{0x502,0x502},{0x504,0x504},{0x506,0x506},{0x508,0x508},
|
||||
{0x50A,0x50A},{0x50C,0x50C},{0x50E,0x50E},{0x510,0x510},{0x512,0x512},{0x514,0x514},
|
||||
{0x516,0x516},{0x518,0x518},{0x51A,0x51A},{0x51C,0x51C},{0x51E,0x51E},{0x520,0x520},
|
||||
{0x522,0x522},{0x524,0x524},{0x526,0x526},{0x528,0x528},{0x52A,0x52A},{0x52C,0x52C},
|
||||
{0x52E,0x52E},{0x531,0x556},{0x10A0,0x10C5},{0x10C7,0x10C7},{0x10CD,0x10CD},{0x13A0,0x13F5},
|
||||
{0x1C90,0x1CBA},{0x1CBD,0x1CBF},{0x1E00,0x1E00},{0x1E02,0x1E02},{0x1E04,0x1E04},{0x1E06,0x1E06},
|
||||
{0x1E08,0x1E08},{0x1E0A,0x1E0A},{0x1E0C,0x1E0C},{0x1E0E,0x1E0E},{0x1E10,0x1E10},{0x1E12,0x1E12},
|
||||
{0x1E14,0x1E14},{0x1E16,0x1E16},{0x1E18,0x1E18},{0x1E1A,0x1E1A},{0x1E1C,0x1E1C},{0x1E1E,0x1E1E},
|
||||
{0x1E20,0x1E20},{0x1E22,0x1E22},{0x1E24,0x1E24},{0x1E26,0x1E26},{0x1E28,0x1E28},{0x1E2A,0x1E2A},
|
||||
{0x1E2C,0x1E2C},{0x1E2E,0x1E2E},{0x1E30,0x1E30},{0x1E32,0x1E32},{0x1E34,0x1E34},{0x1E36,0x1E36},
|
||||
{0x1E38,0x1E38},{0x1E3A,0x1E3A},{0x1E3C,0x1E3C},{0x1E3E,0x1E3E},{0x1E40,0x1E40},{0x1E42,0x1E42},
|
||||
{0x1E44,0x1E44},{0x1E46,0x1E46},{0x1E48,0x1E48},{0x1E4A,0x1E4A},{0x1E4C,0x1E4C},{0x1E4E,0x1E4E},
|
||||
{0x1E50,0x1E50},{0x1E52,0x1E52},{0x1E54,0x1E54},{0x1E56,0x1E56},{0x1E58,0x1E58},{0x1E5A,0x1E5A},
|
||||
{0x1E5C,0x1E5C},{0x1E5E,0x1E5E},{0x1E60,0x1E60},{0x1E62,0x1E62},{0x1E64,0x1E64},{0x1E66,0x1E66},
|
||||
{0x1E68,0x1E68},{0x1E6A,0x1E6A},{0x1E6C,0x1E6C},{0x1E6E,0x1E6E},{0x1E70,0x1E70},{0x1E72,0x1E72},
|
||||
{0x1E74,0x1E74},{0x1E76,0x1E76},{0x1E78,0x1E78},{0x1E7A,0x1E7A},{0x1E7C,0x1E7C},{0x1E7E,0x1E7E},
|
||||
{0x1E80,0x1E80},{0x1E82,0x1E82},{0x1E84,0x1E84},{0x1E86,0x1E86},{0x1E88,0x1E88},{0x1E8A,0x1E8A},
|
||||
{0x1E8C,0x1E8C},{0x1E8E,0x1E8E},{0x1E90,0x1E90},{0x1E92,0x1E92},{0x1E94,0x1E94},{0x1E9E,0x1E9E},
|
||||
{0x1EA0,0x1EA0},{0x1EA2,0x1EA2},{0x1EA4,0x1EA4},{0x1EA6,0x1EA6},{0x1EA8,0x1EA8},{0x1EAA,0x1EAA},
|
||||
{0x1EAC,0x1EAC},{0x1EAE,0x1EAE},{0x1EB0,0x1EB0},{0x1EB2,0x1EB2},{0x1EB4,0x1EB4},{0x1EB6,0x1EB6},
|
||||
{0x1EB8,0x1EB8},{0x1EBA,0x1EBA},{0x1EBC,0x1EBC},{0x1EBE,0x1EBE},{0x1EC0,0x1EC0},{0x1EC2,0x1EC2},
|
||||
{0x1EC4,0x1EC4},{0x1EC6,0x1EC6},{0x1EC8,0x1EC8},{0x1ECA,0x1ECA},{0x1ECC,0x1ECC},{0x1ECE,0x1ECE},
|
||||
{0x1ED0,0x1ED0},{0x1ED2,0x1ED2},{0x1ED4,0x1ED4},{0x1ED6,0x1ED6},{0x1ED8,0x1ED8},{0x1EDA,0x1EDA},
|
||||
{0x1EDC,0x1EDC},{0x1EDE,0x1EDE},{0x1EE0,0x1EE0},{0x1EE2,0x1EE2},{0x1EE4,0x1EE4},{0x1EE6,0x1EE6},
|
||||
{0x1EE8,0x1EE8},{0x1EEA,0x1EEA},{0x1EEC,0x1EEC},{0x1EEE,0x1EEE},{0x1EF0,0x1EF0},{0x1EF2,0x1EF2},
|
||||
{0x1EF4,0x1EF4},{0x1EF6,0x1EF6},{0x1EF8,0x1EF8},{0x1EFA,0x1EFA},{0x1EFC,0x1EFC},{0x1EFE,0x1EFE},
|
||||
{0x1F08,0x1F0F},{0x1F18,0x1F1D},{0x1F28,0x1F2F},{0x1F38,0x1F3F},{0x1F48,0x1F4D},{0x1F59,0x1F59},
|
||||
{0x1F5B,0x1F5B},{0x1F5D,0x1F5D},{0x1F5F,0x1F5F},{0x1F68,0x1F6F},{0x1F88,0x1F8F},{0x1F98,0x1F9F},
|
||||
{0x1FA8,0x1FAF},{0x1FB8,0x1FBC},{0x1FC8,0x1FCC},{0x1FD8,0x1FDB},{0x1FE8,0x1FEC},{0x1FF8,0x1FFC},
|
||||
{0x2102,0x2102},{0x2107,0x2107},{0x210B,0x210D},{0x2110,0x2112},{0x2115,0x2115},{0x2119,0x211D},
|
||||
{0x2124,0x2124},{0x2126,0x2126},{0x2128,0x2128},{0x212A,0x212D},{0x2130,0x2133},{0x213E,0x213F},
|
||||
{0x2145,0x2145},{0x2183,0x2183},{0x2C00,0x2C2E},{0x2C60,0x2C60},{0x2C62,0x2C64},{0x2C67,0x2C67},
|
||||
{0x2C69,0x2C69},{0x2C6B,0x2C6B},{0x2C6D,0x2C70},{0x2C72,0x2C72},{0x2C75,0x2C75},{0x2C7E,0x2C80},
|
||||
{0x2C82,0x2C82},{0x2C84,0x2C84},{0x2C86,0x2C86},{0x2C88,0x2C88},{0x2C8A,0x2C8A},{0x2C8C,0x2C8C},
|
||||
{0x2C8E,0x2C8E},{0x2C90,0x2C90},{0x2C92,0x2C92},{0x2C94,0x2C94},{0x2C96,0x2C96},{0x2C98,0x2C98},
|
||||
{0x2C9A,0x2C9A},{0x2C9C,0x2C9C},{0x2C9E,0x2C9E},{0x2CA0,0x2CA0},{0x2CA2,0x2CA2},{0x2CA4,0x2CA4},
|
||||
{0x2CA6,0x2CA6},{0x2CA8,0x2CA8},{0x2CAA,0x2CAA},{0x2CAC,0x2CAC},{0x2CAE,0x2CAE},{0x2CB0,0x2CB0},
|
||||
{0x2CB2,0x2CB2},{0x2CB4,0x2CB4},{0x2CB6,0x2CB6},{0x2CB8,0x2CB8},{0x2CBA,0x2CBA},{0x2CBC,0x2CBC},
|
||||
{0x2CBE,0x2CBE},{0x2CC0,0x2CC0},{0x2CC2,0x2CC2},{0x2CC4,0x2CC4},{0x2CC6,0x2CC6},{0x2CC8,0x2CC8},
|
||||
{0x2CCA,0x2CCA},{0x2CCC,0x2CCC},{0x2CCE,0x2CCE},{0x2CD0,0x2CD0},{0x2CD2,0x2CD2},{0x2CD4,0x2CD4},
|
||||
{0x2CD6,0x2CD6},{0x2CD8,0x2CD8},{0x2CDA,0x2CDA},{0x2CDC,0x2CDC},{0x2CDE,0x2CDE},{0x2CE0,0x2CE0},
|
||||
{0x2CE2,0x2CE2},{0x2CEB,0x2CEB},{0x2CED,0x2CED},{0x2CF2,0x2CF2},{0xA640,0xA640},{0xA642,0xA642},
|
||||
{0xA644,0xA644},{0xA646,0xA646},{0xA648,0xA648},{0xA64A,0xA64A},{0xA64C,0xA64C},{0xA64E,0xA64E},
|
||||
{0xA650,0xA650},{0xA652,0xA652},{0xA654,0xA654},{0xA656,0xA656},{0xA658,0xA658},{0xA65A,0xA65A},
|
||||
{0xA65C,0xA65C},{0xA65E,0xA65E},{0xA660,0xA660},{0xA662,0xA662},{0xA664,0xA664},{0xA666,0xA666},
|
||||
{0xA668,0xA668},{0xA66A,0xA66A},{0xA66C,0xA66C},{0xA680,0xA680},{0xA682,0xA682},{0xA684,0xA684},
|
||||
{0xA686,0xA686},{0xA688,0xA688},{0xA68A,0xA68A},{0xA68C,0xA68C},{0xA68E,0xA68E},{0xA690,0xA690},
|
||||
{0xA692,0xA692},{0xA694,0xA694},{0xA696,0xA696},{0xA698,0xA698},{0xA69A,0xA69A},{0xA722,0xA722},
|
||||
{0xA724,0xA724},{0xA726,0xA726},{0xA728,0xA728},{0xA72A,0xA72A},{0xA72C,0xA72C},{0xA72E,0xA72E},
|
||||
{0xA732,0xA732},{0xA734,0xA734},{0xA736,0xA736},{0xA738,0xA738},{0xA73A,0xA73A},{0xA73C,0xA73C},
|
||||
{0xA73E,0xA73E},{0xA740,0xA740},{0xA742,0xA742},{0xA744,0xA744},{0xA746,0xA746},{0xA748,0xA748},
|
||||
{0xA74A,0xA74A},{0xA74C,0xA74C},{0xA74E,0xA74E},{0xA750,0xA750},{0xA752,0xA752},{0xA754,0xA754},
|
||||
{0xA756,0xA756},{0xA758,0xA758},{0xA75A,0xA75A},{0xA75C,0xA75C},{0xA75E,0xA75E},{0xA760,0xA760},
|
||||
{0xA762,0xA762},{0xA764,0xA764},{0xA766,0xA766},{0xA768,0xA768},{0xA76A,0xA76A},{0xA76C,0xA76C},
|
||||
{0xA76E,0xA76E},{0xA779,0xA779},{0xA77B,0xA77B},{0xA77D,0xA77E},{0xA780,0xA780},{0xA782,0xA782},
|
||||
{0xA784,0xA784},{0xA786,0xA786},{0xA78B,0xA78B},{0xA78D,0xA78D},{0xA790,0xA790},{0xA792,0xA792},
|
||||
{0xA796,0xA796},{0xA798,0xA798},{0xA79A,0xA79A},{0xA79C,0xA79C},{0xA79E,0xA79E},{0xA7A0,0xA7A0},
|
||||
{0xA7A2,0xA7A2},{0xA7A4,0xA7A4},{0xA7A6,0xA7A6},{0xA7A8,0xA7A8},{0xA7AA,0xA7AE},{0xA7B0,0xA7B4},
|
||||
{0xA7B6,0xA7B6},{0xA7B8,0xA7B8},{0xA7BA,0xA7BA},{0xA7BC,0xA7BC},{0xA7BE,0xA7BE},{0xA7C2,0xA7C2},
|
||||
{0xA7C4,0xA7C7},{0xA7C9,0xA7C9},{0xA7F5,0xA7F5},{0xFF21,0xFF3A},{0x10400,0x10427},{0x104B0,0x104D3},
|
||||
{0x10C80,0x10CB2},{0x118A0,0x118BF},{0x16E40,0x16E5F},{0x1D400,0x1D419},{0x1D434,0x1D44D},{0x1D468,0x1D481},
|
||||
{0x1D49C,0x1D49C},{0x1D49E,0x1D49F},{0x1D4A2,0x1D4A2},{0x1D4A5,0x1D4A6},{0x1D4A9,0x1D4AC},{0x1D4AE,0x1D4B5},
|
||||
{0x1D4D0,0x1D4E9},{0x1D504,0x1D505},{0x1D507,0x1D50A},{0x1D50D,0x1D514},{0x1D516,0x1D51C},{0x1D538,0x1D539},
|
||||
{0x1D53B,0x1D53E},{0x1D540,0x1D544},{0x1D546,0x1D546},{0x1D54A,0x1D550},{0x1D56C,0x1D585},{0x1D5A0,0x1D5B9},
|
||||
{0x1D5D4,0x1D5ED},{0x1D608,0x1D621},{0x1D63C,0x1D655},{0x1D670,0x1D689},{0x1D6A8,0x1D6C0},{0x1D6E2,0x1D6FA},
|
||||
{0x1D71C,0x1D734},{0x1D756,0x1D76E},{0x1D790,0x1D7A8},{0x1D7CA,0x1D7CA},{0x1E900,0x1E921},
|
||||
};
|
||||
static const int uni_U_n = 641;
|
||||
|
||||
static const uint32_t uni_X[][2] = {
|
||||
{0xAA,0xAA},{0xBA,0xBA},{0x1BB,0x1BB},{0x1C0,0x1C3},{0x294,0x294},{0x2B0,0x2C1},
|
||||
{0x2C6,0x2D1},{0x2E0,0x2E4},{0x2EC,0x2EC},{0x2EE,0x2EE},{0x300,0x36F},{0x374,0x374},
|
||||
{0x37A,0x37A},{0x483,0x489},{0x559,0x559},{0x591,0x5BD},{0x5BF,0x5BF},{0x5C1,0x5C2},
|
||||
{0x5C4,0x5C5},{0x5C7,0x5C7},{0x5D0,0x5EA},{0x5EF,0x5F2},{0x610,0x61A},{0x620,0x65F},
|
||||
{0x66E,0x6D3},{0x6D5,0x6DC},{0x6DF,0x6E8},{0x6EA,0x6EF},{0x6FA,0x6FC},{0x6FF,0x6FF},
|
||||
{0x710,0x74A},{0x74D,0x7B1},{0x7CA,0x7F5},{0x7FA,0x7FA},{0x7FD,0x7FD},{0x800,0x82D},
|
||||
{0x840,0x85B},{0x860,0x86A},{0x8A0,0x8B4},{0x8B6,0x8C7},{0x8D3,0x8E1},{0x8E3,0x963},
|
||||
{0x971,0x983},{0x985,0x98C},{0x98F,0x990},{0x993,0x9A8},{0x9AA,0x9B0},{0x9B2,0x9B2},
|
||||
{0x9B6,0x9B9},{0x9BC,0x9C4},{0x9C7,0x9C8},{0x9CB,0x9CE},{0x9D7,0x9D7},{0x9DC,0x9DD},
|
||||
{0x9DF,0x9E3},{0x9F0,0x9F1},{0x9FC,0x9FC},{0x9FE,0x9FE},{0xA01,0xA03},{0xA05,0xA0A},
|
||||
{0xA0F,0xA10},{0xA13,0xA28},{0xA2A,0xA30},{0xA32,0xA33},{0xA35,0xA36},{0xA38,0xA39},
|
||||
{0xA3C,0xA3C},{0xA3E,0xA42},{0xA47,0xA48},{0xA4B,0xA4D},{0xA51,0xA51},{0xA59,0xA5C},
|
||||
{0xA5E,0xA5E},{0xA70,0xA75},{0xA81,0xA83},{0xA85,0xA8D},{0xA8F,0xA91},{0xA93,0xAA8},
|
||||
{0xAAA,0xAB0},{0xAB2,0xAB3},{0xAB5,0xAB9},{0xABC,0xAC5},{0xAC7,0xAC9},{0xACB,0xACD},
|
||||
{0xAD0,0xAD0},{0xAE0,0xAE3},{0xAF9,0xAFF},{0xB01,0xB03},{0xB05,0xB0C},{0xB0F,0xB10},
|
||||
{0xB13,0xB28},{0xB2A,0xB30},{0xB32,0xB33},{0xB35,0xB39},{0xB3C,0xB44},{0xB47,0xB48},
|
||||
{0xB4B,0xB4D},{0xB55,0xB57},{0xB5C,0xB5D},{0xB5F,0xB63},{0xB71,0xB71},{0xB82,0xB83},
|
||||
{0xB85,0xB8A},{0xB8E,0xB90},{0xB92,0xB95},{0xB99,0xB9A},{0xB9C,0xB9C},{0xB9E,0xB9F},
|
||||
{0xBA3,0xBA4},{0xBA8,0xBAA},{0xBAE,0xBB9},{0xBBE,0xBC2},{0xBC6,0xBC8},{0xBCA,0xBCD},
|
||||
{0xBD0,0xBD0},{0xBD7,0xBD7},{0xC00,0xC0C},{0xC0E,0xC10},{0xC12,0xC28},{0xC2A,0xC39},
|
||||
{0xC3D,0xC44},{0xC46,0xC48},{0xC4A,0xC4D},{0xC55,0xC56},{0xC58,0xC5A},{0xC60,0xC63},
|
||||
{0xC80,0xC83},{0xC85,0xC8C},{0xC8E,0xC90},{0xC92,0xCA8},{0xCAA,0xCB3},{0xCB5,0xCB9},
|
||||
{0xCBC,0xCC4},{0xCC6,0xCC8},{0xCCA,0xCCD},{0xCD5,0xCD6},{0xCDE,0xCDE},{0xCE0,0xCE3},
|
||||
{0xCF1,0xCF2},{0xD00,0xD0C},{0xD0E,0xD10},{0xD12,0xD44},{0xD46,0xD48},{0xD4A,0xD4E},
|
||||
{0xD54,0xD57},{0xD5F,0xD63},{0xD7A,0xD7F},{0xD81,0xD83},{0xD85,0xD96},{0xD9A,0xDB1},
|
||||
{0xDB3,0xDBB},{0xDBD,0xDBD},{0xDC0,0xDC6},{0xDCA,0xDCA},{0xDCF,0xDD4},{0xDD6,0xDD6},
|
||||
{0xDD8,0xDDF},{0xDF2,0xDF3},{0xE01,0xE3A},{0xE40,0xE4E},{0xE81,0xE82},{0xE84,0xE84},
|
||||
{0xE86,0xE8A},{0xE8C,0xEA3},{0xEA5,0xEA5},{0xEA7,0xEBD},{0xEC0,0xEC4},{0xEC6,0xEC6},
|
||||
{0xEC8,0xECD},{0xEDC,0xEDF},{0xF00,0xF00},{0xF18,0xF19},{0xF35,0xF35},{0xF37,0xF37},
|
||||
{0xF39,0xF39},{0xF3E,0xF47},{0xF49,0xF6C},{0xF71,0xF84},{0xF86,0xF97},{0xF99,0xFBC},
|
||||
{0xFC6,0xFC6},{0x1000,0x103F},{0x1050,0x108F},{0x109A,0x109D},{0x10FC,0x10FC},{0x1100,0x1248},
|
||||
{0x124A,0x124D},{0x1250,0x1256},{0x1258,0x1258},{0x125A,0x125D},{0x1260,0x1288},{0x128A,0x128D},
|
||||
{0x1290,0x12B0},{0x12B2,0x12B5},{0x12B8,0x12BE},{0x12C0,0x12C0},{0x12C2,0x12C5},{0x12C8,0x12D6},
|
||||
{0x12D8,0x1310},{0x1312,0x1315},{0x1318,0x135A},{0x135D,0x135F},{0x1380,0x138F},{0x1401,0x166C},
|
||||
{0x166F,0x167F},{0x1681,0x169A},{0x16A0,0x16EA},{0x16F1,0x16F8},{0x1700,0x170C},{0x170E,0x1714},
|
||||
{0x1720,0x1734},{0x1740,0x1753},{0x1760,0x176C},{0x176E,0x1770},{0x1772,0x1773},{0x1780,0x17D3},
|
||||
{0x17D7,0x17D7},{0x17DC,0x17DD},{0x180B,0x180D},{0x1820,0x1878},{0x1880,0x18AA},{0x18B0,0x18F5},
|
||||
{0x1900,0x191E},{0x1920,0x192B},{0x1930,0x193B},{0x1950,0x196D},{0x1970,0x1974},{0x1980,0x19AB},
|
||||
{0x19B0,0x19C9},{0x1A00,0x1A1B},{0x1A20,0x1A5E},{0x1A60,0x1A7C},{0x1A7F,0x1A7F},{0x1AA7,0x1AA7},
|
||||
{0x1AB0,0x1AC0},{0x1B00,0x1B4B},{0x1B6B,0x1B73},{0x1B80,0x1BAF},{0x1BBA,0x1BF3},{0x1C00,0x1C37},
|
||||
{0x1C4D,0x1C4F},{0x1C5A,0x1C7D},{0x1CD0,0x1CD2},{0x1CD4,0x1CFA},{0x1D2C,0x1D6A},{0x1D78,0x1D78},
|
||||
{0x1D9B,0x1DF9},{0x1DFB,0x1DFF},{0x2071,0x2071},{0x207F,0x207F},{0x2090,0x209C},{0x20D0,0x20F0},
|
||||
{0x2135,0x2138},{0x2C7C,0x2C7D},{0x2CEF,0x2CF1},{0x2D30,0x2D67},{0x2D6F,0x2D6F},{0x2D7F,0x2D96},
|
||||
{0x2DA0,0x2DA6},{0x2DA8,0x2DAE},{0x2DB0,0x2DB6},{0x2DB8,0x2DBE},{0x2DC0,0x2DC6},{0x2DC8,0x2DCE},
|
||||
{0x2DD0,0x2DD6},{0x2DD8,0x2DDE},{0x2DE0,0x2DFF},{0x2E2F,0x2E2F},{0x3005,0x3006},{0x302A,0x302F},
|
||||
{0x3031,0x3035},{0x303B,0x303C},{0x3041,0x3096},{0x3099,0x309A},{0x309D,0x309F},{0x30A1,0x30FA},
|
||||
{0x30FC,0x30FF},{0x3105,0x312F},{0x3131,0x318E},{0x31A0,0x31BF},{0x31F0,0x31FF},{0x3400,0x4DBF},
|
||||
{0x4E00,0x9FFC},{0xA000,0xA48C},{0xA4D0,0xA4FD},{0xA500,0xA60C},{0xA610,0xA61F},{0xA62A,0xA62B},
|
||||
{0xA66E,0xA672},{0xA674,0xA67D},{0xA67F,0xA67F},{0xA69C,0xA6E5},{0xA6F0,0xA6F1},{0xA717,0xA71F},
|
||||
{0xA770,0xA770},{0xA788,0xA788},{0xA78F,0xA78F},{0xA7F7,0xA7F9},{0xA7FB,0xA827},{0xA82C,0xA82C},
|
||||
{0xA840,0xA873},{0xA880,0xA8C5},{0xA8E0,0xA8F7},{0xA8FB,0xA8FB},{0xA8FD,0xA8FF},{0xA90A,0xA92D},
|
||||
{0xA930,0xA953},{0xA960,0xA97C},{0xA980,0xA9C0},{0xA9CF,0xA9CF},{0xA9E0,0xA9EF},{0xA9FA,0xA9FE},
|
||||
{0xAA00,0xAA36},{0xAA40,0xAA4D},{0xAA60,0xAA76},{0xAA7A,0xAAC2},{0xAADB,0xAADD},{0xAAE0,0xAAEF},
|
||||
{0xAAF2,0xAAF6},{0xAB01,0xAB06},{0xAB09,0xAB0E},{0xAB11,0xAB16},{0xAB20,0xAB26},{0xAB28,0xAB2E},
|
||||
{0xAB5C,0xAB5F},{0xAB69,0xAB69},{0xABC0,0xABEA},{0xABEC,0xABED},{0xAC00,0xD7A3},{0xD7B0,0xD7C6},
|
||||
{0xD7CB,0xD7FB},{0xF900,0xFA6D},{0xFA70,0xFAD9},{0xFB1D,0xFB28},{0xFB2A,0xFB36},{0xFB38,0xFB3C},
|
||||
{0xFB3E,0xFB3E},{0xFB40,0xFB41},{0xFB43,0xFB44},{0xFB46,0xFBB1},{0xFBD3,0xFD3D},{0xFD50,0xFD8F},
|
||||
{0xFD92,0xFDC7},{0xFDF0,0xFDFB},{0xFE00,0xFE0F},{0xFE20,0xFE2F},{0xFE70,0xFE74},{0xFE76,0xFEFC},
|
||||
{0xFF66,0xFFBE},{0xFFC2,0xFFC7},{0xFFCA,0xFFCF},{0xFFD2,0xFFD7},{0xFFDA,0xFFDC},{0x10000,0x1000B},
|
||||
{0x1000D,0x10026},{0x10028,0x1003A},{0x1003C,0x1003D},{0x1003F,0x1004D},{0x10050,0x1005D},{0x10080,0x100FA},
|
||||
{0x101FD,0x101FD},{0x10280,0x1029C},{0x102A0,0x102D0},{0x102E0,0x102E0},{0x10300,0x1031F},{0x1032D,0x10340},
|
||||
{0x10342,0x10349},{0x10350,0x1037A},{0x10380,0x1039D},{0x103A0,0x103C3},{0x103C8,0x103CF},{0x10450,0x1049D},
|
||||
{0x10500,0x10527},{0x10530,0x10563},{0x10600,0x10736},{0x10740,0x10755},{0x10760,0x10767},{0x10800,0x10805},
|
||||
{0x10808,0x10808},{0x1080A,0x10835},{0x10837,0x10838},{0x1083C,0x1083C},{0x1083F,0x10855},{0x10860,0x10876},
|
||||
{0x10880,0x1089E},{0x108E0,0x108F2},{0x108F4,0x108F5},{0x10900,0x10915},{0x10920,0x10939},{0x10980,0x109B7},
|
||||
{0x109BE,0x109BF},{0x10A00,0x10A03},{0x10A05,0x10A06},{0x10A0C,0x10A13},{0x10A15,0x10A17},{0x10A19,0x10A35},
|
||||
{0x10A38,0x10A3A},{0x10A3F,0x10A3F},{0x10A60,0x10A7C},{0x10A80,0x10A9C},{0x10AC0,0x10AC7},{0x10AC9,0x10AE6},
|
||||
{0x10B00,0x10B35},{0x10B40,0x10B55},{0x10B60,0x10B72},{0x10B80,0x10B91},{0x10C00,0x10C48},{0x10D00,0x10D27},
|
||||
{0x10E80,0x10EA9},{0x10EAB,0x10EAC},{0x10EB0,0x10EB1},{0x10F00,0x10F1C},{0x10F27,0x10F27},{0x10F30,0x10F50},
|
||||
{0x10FB0,0x10FC4},{0x10FE0,0x10FF6},{0x11000,0x11046},{0x1107F,0x110BA},{0x110D0,0x110E8},{0x11100,0x11134},
|
||||
{0x11144,0x11147},{0x11150,0x11173},{0x11176,0x11176},{0x11180,0x111C4},{0x111C9,0x111CC},{0x111CE,0x111CF},
|
||||
{0x111DA,0x111DA},{0x111DC,0x111DC},{0x11200,0x11211},{0x11213,0x11237},{0x1123E,0x1123E},{0x11280,0x11286},
|
||||
{0x11288,0x11288},{0x1128A,0x1128D},{0x1128F,0x1129D},{0x1129F,0x112A8},{0x112B0,0x112EA},{0x11300,0x11303},
|
||||
{0x11305,0x1130C},{0x1130F,0x11310},{0x11313,0x11328},{0x1132A,0x11330},{0x11332,0x11333},{0x11335,0x11339},
|
||||
{0x1133B,0x11344},{0x11347,0x11348},{0x1134B,0x1134D},{0x11350,0x11350},{0x11357,0x11357},{0x1135D,0x11363},
|
||||
{0x11366,0x1136C},{0x11370,0x11374},{0x11400,0x1144A},{0x1145E,0x11461},{0x11480,0x114C5},{0x114C7,0x114C7},
|
||||
{0x11580,0x115B5},{0x115B8,0x115C0},{0x115D8,0x115DD},{0x11600,0x11640},{0x11644,0x11644},{0x11680,0x116B8},
|
||||
{0x11700,0x1171A},{0x1171D,0x1172B},{0x11800,0x1183A},{0x118FF,0x11906},{0x11909,0x11909},{0x1190C,0x11913},
|
||||
{0x11915,0x11916},{0x11918,0x11935},{0x11937,0x11938},{0x1193B,0x11943},{0x119A0,0x119A7},{0x119AA,0x119D7},
|
||||
{0x119DA,0x119E1},{0x119E3,0x119E4},{0x11A00,0x11A3E},{0x11A47,0x11A47},{0x11A50,0x11A99},{0x11A9D,0x11A9D},
|
||||
{0x11AC0,0x11AF8},{0x11C00,0x11C08},{0x11C0A,0x11C36},{0x11C38,0x11C40},{0x11C72,0x11C8F},{0x11C92,0x11CA7},
|
||||
{0x11CA9,0x11CB6},{0x11D00,0x11D06},{0x11D08,0x11D09},{0x11D0B,0x11D36},{0x11D3A,0x11D3A},{0x11D3C,0x11D3D},
|
||||
{0x11D3F,0x11D47},{0x11D60,0x11D65},{0x11D67,0x11D68},{0x11D6A,0x11D8E},{0x11D90,0x11D91},{0x11D93,0x11D98},
|
||||
{0x11EE0,0x11EF6},{0x11FB0,0x11FB0},{0x12000,0x12399},{0x12480,0x12543},{0x13000,0x1342E},{0x14400,0x14646},
|
||||
{0x16800,0x16A38},{0x16A40,0x16A5E},{0x16AD0,0x16AED},{0x16AF0,0x16AF4},{0x16B00,0x16B36},{0x16B40,0x16B43},
|
||||
{0x16B63,0x16B77},{0x16B7D,0x16B8F},{0x16F00,0x16F4A},{0x16F4F,0x16F87},{0x16F8F,0x16F9F},{0x16FE0,0x16FE1},
|
||||
{0x16FE3,0x16FE4},{0x16FF0,0x16FF1},{0x17000,0x187F7},{0x18800,0x18CD5},{0x18D00,0x18D08},{0x1B000,0x1B11E},
|
||||
{0x1B150,0x1B152},{0x1B164,0x1B167},{0x1B170,0x1B2FB},{0x1BC00,0x1BC6A},{0x1BC70,0x1BC7C},{0x1BC80,0x1BC88},
|
||||
{0x1BC90,0x1BC99},{0x1BC9D,0x1BC9E},{0x1D165,0x1D169},{0x1D16D,0x1D172},{0x1D17B,0x1D182},{0x1D185,0x1D18B},
|
||||
{0x1D1AA,0x1D1AD},{0x1D242,0x1D244},{0x1DA00,0x1DA36},{0x1DA3B,0x1DA6C},{0x1DA75,0x1DA75},{0x1DA84,0x1DA84},
|
||||
{0x1DA9B,0x1DA9F},{0x1DAA1,0x1DAAF},{0x1E000,0x1E006},{0x1E008,0x1E018},{0x1E01B,0x1E021},{0x1E023,0x1E024},
|
||||
{0x1E026,0x1E02A},{0x1E100,0x1E12C},{0x1E130,0x1E13D},{0x1E14E,0x1E14E},{0x1E2C0,0x1E2EF},{0x1E800,0x1E8C4},
|
||||
{0x1E8D0,0x1E8D6},{0x1E944,0x1E94B},{0x1EE00,0x1EE03},{0x1EE05,0x1EE1F},{0x1EE21,0x1EE22},{0x1EE24,0x1EE24},
|
||||
{0x1EE27,0x1EE27},{0x1EE29,0x1EE32},{0x1EE34,0x1EE37},{0x1EE39,0x1EE39},{0x1EE3B,0x1EE3B},{0x1EE42,0x1EE42},
|
||||
{0x1EE47,0x1EE47},{0x1EE49,0x1EE49},{0x1EE4B,0x1EE4B},{0x1EE4D,0x1EE4F},{0x1EE51,0x1EE52},{0x1EE54,0x1EE54},
|
||||
{0x1EE57,0x1EE57},{0x1EE59,0x1EE59},{0x1EE5B,0x1EE5B},{0x1EE5D,0x1EE5D},{0x1EE5F,0x1EE5F},{0x1EE61,0x1EE62},
|
||||
{0x1EE64,0x1EE64},{0x1EE67,0x1EE6A},{0x1EE6C,0x1EE72},{0x1EE74,0x1EE77},{0x1EE79,0x1EE7C},{0x1EE7E,0x1EE7E},
|
||||
{0x1EE80,0x1EE89},{0x1EE8B,0x1EE9B},{0x1EEA1,0x1EEA3},{0x1EEA5,0x1EEA9},{0x1EEAB,0x1EEBB},{0x20000,0x2A6DD},
|
||||
{0x2A700,0x2B734},{0x2B740,0x2B81D},{0x2B820,0x2CEA1},{0x2CEB0,0x2EBE0},{0x2F800,0x2FA1D},{0x30000,0x3134A},
|
||||
{0xE0100,0xE01EF},
|
||||
};
|
||||
static const int uni_X_n = 595;
|
||||
|
||||
static inline int is_U(uint32_t c){ return uni_in(uni_U,uni_U_n,c); }
|
||||
static inline int is_X(uint32_t c){ return uni_in(uni_X,uni_X_n,c); }
|
||||
#endif
|
||||
@@ -83,6 +83,29 @@ def quant_int4_grouped(w, bits, gs=128):
|
||||
s_flat = s[:, :, 0].astype(np.float32).reshape(-1)
|
||||
return out.reshape(-1), s_flat
|
||||
|
||||
def quant_int3_g64(w, bits=3, group=64): # -> (qbytes U8 [O*ceil(I/64)*24], scales f32 [O*ceil(I/64)])
|
||||
"""int3 with PER-GROUP scales (fmt=5 in colibri.c): per 64-input group, symmetric absmax
|
||||
(qmax=3, clamp [-4,3], stored v+4), packed as 16B low plane (2 bits/val, int2 layout)
|
||||
+ 8B high plane (1 bit/val). Same math as quant_ablation._quant_last_dim(bits=3,
|
||||
group=64) (#132), here with real packing. 3.5 bits/weight effective."""
|
||||
O, I = w.shape
|
||||
ng = (I + group - 1) // group
|
||||
pad = ng * group - I
|
||||
wp = np.pad(w, ((0, 0), (0, pad))) if pad else w
|
||||
g = wp.reshape(O, ng, group)
|
||||
amax = np.abs(g).max(axis=2, keepdims=True)
|
||||
s = np.maximum(amax / 3.0, 1e-8)
|
||||
q = (np.clip(np.rint(g / s), -4, 3).astype(np.int32) + 4).astype(np.uint8) # 0..7
|
||||
if pad: q[:, -1, group - pad:] = 4 # pad packs as 0 after -4
|
||||
lo = np.zeros((O, ng, 16), np.uint8)
|
||||
for k in range(4):
|
||||
lo |= ((q[:, :, k::4] & 3) << (k * 2)).astype(np.uint8)
|
||||
hi = np.zeros((O, ng, 8), np.uint8)
|
||||
for b in range(8):
|
||||
hi |= (((q[:, :, b::8] >> 2) & 1) << b).astype(np.uint8)
|
||||
out = np.concatenate([lo, hi], axis=2) # [O, ng, 24]
|
||||
return out.reshape(-1), s[:, :, 0].astype(np.float32).reshape(-1)
|
||||
|
||||
def quant_int2(w, bits): # -> (qbytes U8 [O*ceil(I/4)], scale f32 [O]); 4/byte
|
||||
O, I = w.shape
|
||||
qmax = (1 << (bits - 1)) - 1 # bits=2 -> qmax=1, valori [-2,1]
|
||||
@@ -214,6 +237,14 @@ def dequant(f, name, keys):
|
||||
return (w * sc).numpy()
|
||||
return f.get_tensor(name).to(torch.float32).numpy()
|
||||
|
||||
# Per-projection bit overrides for ROUTED experts (gate_proj/up_proj/down_proj), set from
|
||||
# --up-bits/--gate-bits/--down-bits in main(). Empty = uniform xbits. Motivated by the
|
||||
# measured result that up_proj tolerates int3-g64 at ~zero quality cost while int2 craters
|
||||
# (OLMoE ablation, PR #168 comment): up-only int3 drops ~8% of expert bytes for free.
|
||||
# NB: the resume manifests (check_or_record_params and the --indir progress file) already
|
||||
# record dict(PROJ_BITS) — this global is the definition those sites depend on.
|
||||
PROJ_BITS = {}
|
||||
|
||||
def convert_shard(path, out_dict, n_layers, ebits, io_bits, xbits,
|
||||
keep_mtp=False, keep_idx=False, group_size=0, bits_map=None):
|
||||
from safetensors import safe_open
|
||||
@@ -235,9 +266,16 @@ def convert_shard(path, out_dict, n_layers, ebits, io_bits, xbits,
|
||||
# Any unknown kind that fell through classify as "q"
|
||||
if bits_map and kind not in bits_map and kind not in ("io", "x", "sh", "o", "kvb", "attn", "dmlp"):
|
||||
bits = ebits
|
||||
# Per-projection override for routed experts, applied on top of the type-level bits.
|
||||
if kind == "x" and PROJ_BITS: # e.g. up_proj -> 3 (int3-g64) while gate/down stay 4
|
||||
for proj, pb in PROJ_BITS.items():
|
||||
if f".{proj}.weight" in name: bits = pb; break
|
||||
if w.ndim != 2: # es. bias 1D non previsto come 'q' -> tienilo f32
|
||||
out_dict[name] = w.astype(np.float32); continue
|
||||
if group_size > 0 and bits <= 4:
|
||||
if bits == 3:
|
||||
# int3-g64 (fmt=5): inherently group-64, distinct from grouped-int4.
|
||||
q, s = quant_int3_g64(w)
|
||||
elif group_size > 0 and bits <= 4:
|
||||
q, s = quant_int4_grouped(w, bits, group_size)
|
||||
else:
|
||||
q, s = (quant_int2(w, bits) if bits <= 2 else
|
||||
@@ -247,6 +285,32 @@ def convert_shard(path, out_dict, n_layers, ebits, io_bits, xbits,
|
||||
|
||||
def free_gb(p): return shutil.disk_usage(p).free / 1e9
|
||||
|
||||
def check_or_record_params(outdir, prefix, params):
|
||||
"""#383-class guard, mirrored onto the --repo download loops from the --indir
|
||||
path's resume manifest (below): a resumed run with DIFFERENT conversion
|
||||
parameters (bits, group size, PROJ_BITS, ...) must not silently mix bit-widths
|
||||
across shards in the same outdir -- the #355 failure mode (a second pass with
|
||||
changed flags overwriting/interleaving with a finished container in silence).
|
||||
Unlike the --indir manifest this doesn't need to track per-shard completion:
|
||||
the --repo loops already do that via out-NNNNN.safetensors existence, since
|
||||
shard index maps directly to output filename there. Only whether the params
|
||||
used SO FAR match this run's needs checking. Returns False (caller should
|
||||
abort) on a mismatch, True otherwise; records params on first use."""
|
||||
path = os.path.join(outdir, f".{prefix}params.json")
|
||||
if os.path.exists(path):
|
||||
try: prev = json.loads(open(path).read())
|
||||
except (OSError, ValueError): prev = None
|
||||
if prev is not None and prev != params:
|
||||
print(f"ERROR: {path} records a conversion with {prev};\n"
|
||||
f" this run uses {params}. Refusing to mix conversions in the "
|
||||
f"same outdir — use a fresh --outdir (or delete {path} and the "
|
||||
f"{prefix}*.safetensors shards to redo).")
|
||||
return False
|
||||
tmp = path + ".tmp"
|
||||
with open(tmp, "w") as f: json.dump(params, f, indent=1) # atomic write, same reasoning as the --indir manifest
|
||||
os.replace(tmp, path)
|
||||
return True
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--repo", default=None)
|
||||
@@ -269,6 +333,13 @@ def main():
|
||||
help="bits for dense MLP (first 3 layers). Default=ebits")
|
||||
ap.add_argument("--group-size", type=int, default=0, # 0 = per-row (backward compat); 128 = group-scaled
|
||||
help="group size for int4 scales: 0=per-row (default), 128=one scale per 128 elements (much better quality)")
|
||||
# Per-projection bit overrides for routed experts (orthogonal to the type-level flags above).
|
||||
ap.add_argument("--up-bits", type=int, default=None,
|
||||
help="bits for up_proj in routed experts (e.g. 3 = int3-g64). Default=xbits")
|
||||
ap.add_argument("--gate-bits", type=int, default=None,
|
||||
help="bits for gate_proj in routed experts. Default=xbits")
|
||||
ap.add_argument("--down-bits", type=int, default=None,
|
||||
help="bits for down_proj in routed experts. Default=xbits")
|
||||
ap.add_argument("--n-layers", type=int, default=78)
|
||||
ap.add_argument("--min-free-gb", type=float, default=20.0)
|
||||
ap.add_argument("--selftest", action="store_true")
|
||||
@@ -298,6 +369,10 @@ def main():
|
||||
"embedding half -> MTP acceptance ~0% (issue #8). Use the default --ebits 8, "
|
||||
"or add --group-size 128 for group-scaled int4.")
|
||||
if a.xbits is None: a.xbits = a.ebits
|
||||
for proj, val in (("gate_proj", a.gate_bits), ("up_proj", a.up_bits), ("down_proj", a.down_bits)):
|
||||
if val is not None: PROJ_BITS[proj] = val
|
||||
if PROJ_BITS:
|
||||
print(f"[per-projection expert bits] {PROJ_BITS} (others -> xbits={a.xbits})")
|
||||
|
||||
# Build per-type bits map. If a type-specific arg is set, use it; otherwise the
|
||||
# converter falls back to ebits for that type.
|
||||
@@ -440,7 +515,8 @@ def main():
|
||||
# EN: resume skips only what matches, and different parameters on the same
|
||||
# EN: outdir are refused instead of mixing containers (the #355 failure mode).
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map}
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
prog_path = os.path.join(a.outdir, f".{prefix}progress.json")
|
||||
prog = {}
|
||||
if os.path.exists(prog_path):
|
||||
@@ -718,6 +794,10 @@ def main():
|
||||
except Exception: pass
|
||||
tmp = os.path.join(a.outdir, "_inflight"); os.makedirs(tmp, exist_ok=True)
|
||||
if a.mtp:
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-mtp-", params): return
|
||||
import urllib.request
|
||||
idx = json.loads(urllib.request.urlopen(
|
||||
f"https://huggingface.co/{a.repo}/resolve/main/model.safetensors.index.json", timeout=30).read())["weight_map"]
|
||||
@@ -737,6 +817,10 @@ def main():
|
||||
print(f" -> {os.path.basename(outp)} ({os.path.getsize(outp)/1e9:.2f} GB, {len(out)} tensors)", flush=True)
|
||||
shutil.rmtree(tmp, ignore_errors=True); print("[MTP] DONE."); return
|
||||
if a.indexer:
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-idx-", params): return
|
||||
import urllib.request
|
||||
idx = json.loads(urllib.request.urlopen(
|
||||
f"https://huggingface.co/{a.repo}/resolve/main/model.safetensors.index.json", timeout=30).read())["weight_map"]
|
||||
@@ -756,6 +840,10 @@ def main():
|
||||
if os.path.isfile(blob): os.remove(blob)
|
||||
print(f" -> {os.path.basename(outp)} ({len(out)} tensors)", flush=True)
|
||||
shutil.rmtree(tmp, ignore_errors=True); print("[IDX] DONE."); return
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-", params): return
|
||||
for i, sh in enumerate(shards):
|
||||
if free_gb(a.outdir) < a.min_free_gb:
|
||||
print(f"STOP: free space is below {a.min_free_gb} GB. Free space and rerun to resume."); break
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert OLMoE HuggingFace checkpoint to colibri merged int8 format.
|
||||
|
||||
Consolidates gate_proj, up_proj, and down_proj into a single merged tensor per expert.
|
||||
This allows olmoe.c to load an expert in a single disk read call instead of 3.
|
||||
|
||||
Usage:
|
||||
python tools/convert_olmoe_merged.py --repo allenai/OLMoE-1B-7B-0125-Instruct --out ./olmoe_merged
|
||||
"""
|
||||
|
||||
import argparse, json, os, sys, re
|
||||
from pathlib import Path
|
||||
|
||||
# Windows: force UTF-8 output
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try: s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError): pass
|
||||
|
||||
try:
|
||||
import torch
|
||||
from safetensors.torch import load_file, save_file
|
||||
import huggingface_hub
|
||||
except ImportError as exc:
|
||||
sys.exit(f"Missing dependencies: {exc}. Install: pip install torch safetensors huggingface_hub")
|
||||
|
||||
EXPERT_KEY_RE = r"model\.layers\.(\d+)\.mlp\.experts\.(\d+)\.(gate_proj|up_proj|down_proj)\.weight"
|
||||
|
||||
def quantize_row(w: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
"""Row-wise int8 quantization. Returns (int8_weights, float32_scales)."""
|
||||
w_f32 = w.float()
|
||||
row_max = w_f32.abs().amax(dim=1, keepdim=True).clamp(min=1e-12)
|
||||
scales = row_max / 127.0
|
||||
q = (w_f32 / scales).round().clamp(-128, 127).to(torch.int8)
|
||||
return q, scales.squeeze(1)
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Convert OLMoE HF checkpoint -> colibri merged int8")
|
||||
src = ap.add_mutually_exclusive_group(required=True)
|
||||
src.add_argument("--repo", help="HuggingFace repo ID")
|
||||
src.add_argument("--model", help="Local HF checkpoint directory")
|
||||
ap.add_argument("--out", required=True, help="Output directory for merged model")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.repo:
|
||||
from huggingface_hub import snapshot_download
|
||||
from huggingface_hub.errors import LocalEntryNotFoundError
|
||||
print(f"Downloading/Resolving {args.repo}...")
|
||||
try:
|
||||
src_dir = snapshot_download(args.repo, local_files_only=True, max_workers=4)
|
||||
except LocalEntryNotFoundError:
|
||||
src_dir = None
|
||||
if src_dir is None or not any(Path(src_dir).glob("*.safetensors")):
|
||||
print("Downloading safetensors...")
|
||||
src_dir = snapshot_download(args.repo, max_workers=4)
|
||||
else:
|
||||
src_dir = args.model
|
||||
|
||||
src = Path(src_dir)
|
||||
if not src.is_dir():
|
||||
sys.exit(f"Model directory not found: {src}")
|
||||
if not (src / "config.json").is_file():
|
||||
sys.exit(f"config.json missing in {src}")
|
||||
|
||||
out = Path(args.out)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Copy config.json
|
||||
import shutil
|
||||
shutil.copy2(src / "config.json", out / "config.json")
|
||||
print(f"config.json -> {out}")
|
||||
|
||||
# Process safetensors
|
||||
shards = sorted(src.glob("*.safetensors"))
|
||||
if not shards:
|
||||
sys.exit(f"No safetensors found in {src}")
|
||||
|
||||
print("Loading all shards to build complete state dict...")
|
||||
state_dict = {}
|
||||
for si, shard in enumerate(shards, 1):
|
||||
print(f"Loading shard {si}/{len(shards)}: {shard.name}...")
|
||||
tensors = load_file(str(shard))
|
||||
state_dict.update(tensors)
|
||||
|
||||
# Gather experts
|
||||
experts = {}
|
||||
for name in list(state_dict.keys()):
|
||||
m = re.match(EXPERT_KEY_RE, name)
|
||||
if m:
|
||||
layer_idx, expert_idx, proj = m.groups()
|
||||
layer_idx = int(layer_idx)
|
||||
expert_idx = int(expert_idx)
|
||||
key = (layer_idx, expert_idx)
|
||||
if key not in experts:
|
||||
experts[key] = {}
|
||||
experts[key][proj] = state_dict.pop(name)
|
||||
|
||||
print(f"Found {len(experts)} experts to merge.")
|
||||
|
||||
# Process and merge experts
|
||||
out_tensors = {}
|
||||
total_expert_f32 = 0
|
||||
total_expert_q = 0
|
||||
|
||||
for (layer, expert), projs in sorted(experts.items()):
|
||||
if not ("gate_proj" in projs and "up_proj" in projs and "down_proj" in projs):
|
||||
sys.exit(f"Missing projection for layer {layer} expert {expert}!")
|
||||
|
||||
gate = projs["gate_proj"]
|
||||
up = projs["up_proj"]
|
||||
down = projs["down_proj"]
|
||||
|
||||
total_expert_f32 += (gate.numel() + up.numel() + down.numel()) * gate.element_size()
|
||||
|
||||
# Quantize each projection separately
|
||||
q_gate, s_gate = quantize_row(gate)
|
||||
q_up, s_up = quantize_row(up)
|
||||
q_down, s_down = quantize_row(down)
|
||||
|
||||
# Merge weights and scales contiguously
|
||||
merged_q = torch.cat([q_gate.flatten(), q_up.flatten(), q_down.flatten()])
|
||||
merged_scales = torch.cat([s_gate, s_up, s_down])
|
||||
|
||||
total_expert_q += merged_q.numel() * 1 + merged_scales.numel() * 4
|
||||
|
||||
# Save to output
|
||||
out_tensors[f"model.layers.{layer}.mlp.experts.{expert}.merged_weight"] = merged_q
|
||||
out_tensors[f"model.layers.{layer}.mlp.experts.{expert}.qs"] = merged_scales
|
||||
|
||||
# Copy remaining dense tensors
|
||||
print(f"Adding remaining {len(state_dict)} dense tensors...")
|
||||
out_tensors.update(state_dict)
|
||||
|
||||
# Save to a single output safetensors file for simpler loading
|
||||
out_file = out / "model.safetensors"
|
||||
print(f"Saving merged safetensors model to {out_file}...")
|
||||
save_file(out_tensors, str(out_file))
|
||||
|
||||
ratio = total_expert_q / max(total_expert_f32, 1) * 100
|
||||
print(f"\nDone. {len(experts)} experts successfully merged and saved.")
|
||||
print(f"Expert storage: {total_expert_f32/1e9:.1f} GB -> {total_expert_q/1e9:.1f} GB ({ratio:.0f}%)")
|
||||
print(f"Model ready at: {out}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,704 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
diag_harness.py — Comprehensive model diagnostic harness for the colibri GLM-5.2 engine.
|
||||
|
||||
Runs a full campaign of tests against any model snapshot:
|
||||
Phase 0 (system): startup telemetry — GPU, RAM, cache cap, load time, MTP, idot kernel
|
||||
Phase 1 (smoke): correctness — curated prompts, coherence checks, corruption detection
|
||||
Phase 2 (diagnostic): deep x-ray — full PROFILE breakdown, routing, MTP, disk-split, CUDA tier
|
||||
Phase 3 (quality): benchmark accuracy — hellaswag/arc_challenge/mmlu via eval_glm.py SCORE
|
||||
Phase 4 (throughput): tok/s with and without MTP speculation
|
||||
Phase 5 (report): structured JSON + human-readable Markdown summary
|
||||
|
||||
Usage:
|
||||
python tools/diag_harness.py --snap /path/to/model --phase all
|
||||
python tools/diag_harness.py --snap /path/to/model --phase smoke --ngen 64
|
||||
python tools/diag_harness.py --snap /path/to/model --phase quality --quality-limit 200
|
||||
|
||||
Output goes to --out (default ./diag_results/<timestamp>/). Each phase writes a raw log
|
||||
(<phase>_<run>.log) and all metrics are collected into report.json + report.md.
|
||||
|
||||
Design notes:
|
||||
- stdout and stderr are captured separately via subprocess.PIPE. The engine streams
|
||||
generated text to stdout (interleaved with prompt + PROFILE stats), but TOKENS=1 dumps
|
||||
clean token-id lists to stderr — that is the primary text-capture path.
|
||||
- Every regex is anchored to the exact printf format strings in glm.c (verified against
|
||||
profile_print line 3853, run_text line 3948, the banner line 5299, etc.).
|
||||
- Subprocess calls have a hard timeout (default 600s) and are killed cleanly on expiry.
|
||||
- A single phase can be run standalone; results accumulate in the output dir.
|
||||
"""
|
||||
import os, sys, re, json, time, argparse, subprocess, signal, traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PROMPT SUITE — curated across categories. "expect" is a case-insensitive
|
||||
# substring checked against the generated text (None = coherence-only check).
|
||||
# ---------------------------------------------------------------------------
|
||||
PROMPTS = [
|
||||
{"id":"fact_capital", "cat":"factual", "prompt":"The capital of France is",
|
||||
"expect":"Paris", "note":"basic world knowledge"},
|
||||
{"id":"fact_boiling", "cat":"factual", "prompt":"What is the boiling point of water in Celsius?",
|
||||
"expect":"100", "note":"basic science"},
|
||||
{"id":"fact_planet", "cat":"factual", "prompt":"What planet is closest to the Sun?",
|
||||
"expect":"Mercury","note":"astronomy fact"},
|
||||
{"id":"math_mult", "cat":"math", "prompt":"What is 15 times 12?",
|
||||
"expect":"180", "note":"2-digit multiplication"},
|
||||
{"id":"math_add", "cat":"math", "prompt":"What is 847 plus 153?",
|
||||
"expect":"1000", "note":"3-digit addition"},
|
||||
{"id":"reason_train", "cat":"reasoning", "prompt":"If a train travels 60 mph for 2.5 hours, how far does it go?",
|
||||
"expect":"150", "note":"rate-time-distance"},
|
||||
{"id":"code_factorial","cat":"code", "prompt":"Write a Python function that computes the factorial of a number.",
|
||||
"expect":"def", "note":"code generation"},
|
||||
{"id":"explain_nn", "cat":"explanation","prompt":"Explain what a neural network is in one sentence.",
|
||||
"expect":None, "note":"coherence check"},
|
||||
{"id":"creative_story","cat":"creative", "prompt":"Write a one-sentence story about a lighthouse.",
|
||||
"expect":None, "note":"creative coherence"},
|
||||
{"id":"edge_hello", "cat":"edge", "prompt":"Hello, how are you today?",
|
||||
"expect":None, "note":"conversational opener"},
|
||||
{"id":"edge_the", "cat":"edge", "prompt":"The weather today is",
|
||||
"expect":None, "note":"simple continuation"},
|
||||
{"id":"edge_repeat", "cat":"edge", "prompt":"The quick brown fox jumps over the lazy dog. The quick brown fox",
|
||||
"expect":None, "note":"repetition-bait"},
|
||||
]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# METRIC EXTRACTION — each parser is (name, compiled_regex, group_index, cast).
|
||||
# Regexes match the EXACT printf format strings in glm.c.
|
||||
# ---------------------------------------------------------------------------
|
||||
def _f1(g): return float(g)
|
||||
def _i1(g): return int(g)
|
||||
|
||||
# Build parsers as (regex, lambda(match)->value) for clarity
|
||||
RX = {
|
||||
# stdout: run_text summary line (line 3948)
|
||||
"decode_toks": (re.compile(r"decode (\d+) tokens in"), lambda m: int(m.group(1))),
|
||||
"decode_secs": (re.compile(r"decode \d+ tokens in ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"decode_tps": (re.compile(r"decode \d+ tokens in [\d.]+s \(([\d.]+) tok/s\)"), lambda m: float(m.group(1))),
|
||||
"prefill_toks": (re.compile(r"prefill (\d+) tokens in"), lambda m: int(m.group(1))),
|
||||
"prefill_secs": (re.compile(r"prefill \d+ tokens in ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"hit_rate": (re.compile(r"expert hit rate ([\d.]+)%"), lambda m: float(m.group(1))),
|
||||
"rss_gb": (re.compile(r"RSS ([\d.]+) GB"), lambda m: float(m.group(1))),
|
||||
"experts_per_tok":(re.compile(r"experts loaded/token: ([\d.]+)"), lambda m: float(m.group(1))),
|
||||
"mtp_accept": (re.compile(r"MTP acceptance ([\d.]+)%"), lambda m: float(m.group(1))),
|
||||
"mtp_acc_cnt": (re.compile(r"MTP acceptance \d+% \((\d+)/(\d+)\)"), lambda m: (int(m.group(1)), int(m.group(2)))),
|
||||
"spec_tok_per_fw":(re.compile(r"speculation: ([\d.]+) tokens/forward"),lambda m: float(m.group(1))),
|
||||
# stdout: PROFILE lines (profile_print, line 3853-3864)
|
||||
"prof_expert_disk": (re.compile(r"expert-disk ([\d.]+)s service"), lambda m: float(m.group(1))),
|
||||
"prof_expert_wait": (re.compile(r"service / ([\d.]+)s wait"), lambda m: float(m.group(1))),
|
||||
"prof_expert_mm": (re.compile(r"expert-matmul ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"prof_attention": (re.compile(r"\| attention ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"prof_kvb": (re.compile(r"including kvb ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"prof_lm_head": (re.compile(r"lm_head ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"prof_other": (re.compile(r"\| other ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
# stdout: banner (line 5299)
|
||||
"load_secs": (re.compile(r"loaded in ([\d.]+)s"), lambda m: float(m.group(1))),
|
||||
"resident_mb": (re.compile(r"resident dense: ([\d.]+) MB"), lambda m: float(m.group(1))),
|
||||
"mtp_status": (re.compile(r"MTP (ACTIVE|absent)"), lambda m: m.group(1)),
|
||||
"n_layers": (re.compile(r"layers=(\d+) experts=(\d+)"), lambda m: int(m.group(1))),
|
||||
"n_experts": (re.compile(r"layers=(\d+) experts=(\d+)"), lambda m: int(m.group(2))),
|
||||
# stdout: banner idot kernel
|
||||
"idot_kernel": (re.compile(r"idot: (\S+) =="), lambda m: m.group(1)),
|
||||
"cache_cap": (re.compile(r"cache=(\d+) experts/layer"), lambda m: int(m.group(1))),
|
||||
# stderr: RAM_GB (line 5086-5112)
|
||||
"ram_budget": (re.compile(r"\[RAM_GB=([\d.]+)"), lambda m: float(m.group(1))),
|
||||
"cap_lowered": (re.compile(r"cap lowered (\d+)->(\d+)"), lambda m: (int(m.group(1)), int(m.group(2)))),
|
||||
"cap_raised": (re.compile(r"cap raised (\d+)->(\d+)"), lambda m: (int(m.group(1)), int(m.group(2)))),
|
||||
"cap_ok": (re.compile(r"cap=(\d+) ok"), lambda m: int(m.group(1))),
|
||||
# stderr: CUDA (backend_cuda.cu:389)
|
||||
"cuda_device": (re.compile(r"\[CUDA\] device (\d+): (.*?), ([\d.]+) GB VRAM, sm_(\d)(\d)"),
|
||||
lambda m: {"id":int(m.group(1)),"name":m.group(2).strip(),"vram_gb":float(m.group(3)),
|
||||
"sm":f"{m.group(4)}.{m.group(5)}"}),
|
||||
"cuda_mode": (re.compile(r"\[CUDA\] mode: (.+)"), lambda m: m.group(1)),
|
||||
"cuda_tier": (re.compile(r"CUDA expert tier: (\d+) resident experts \(([\d.]+) GB\)"),
|
||||
lambda m: {"resident":int(m.group(1)),"vram_gb":float(m.group(2))}),
|
||||
# stderr: TOKENS dump (line 4010-4012)
|
||||
"tokens_dump": (re.compile(r"^\[TOKENS\] (\d+) generated:(.*)$", re.MULTILINE),
|
||||
lambda m: [int(x) for x in m.group(2).split()]),
|
||||
# stderr: per-16-token progress (emit_stream line 3742)
|
||||
"progress_tps": (re.compile(r"t=(\d+)\s+RSS ([\d.]+) GB\s+hit ([\d.]+)%\s+([\d.]+) tok/s\s+([\d.]+) tok/fw"),
|
||||
lambda m: {"tok":int(m.group(1)),"rss":float(m.group(2)),"hit":float(m.group(3)),
|
||||
"tps":float(m.group(4)),"tpf":float(m.group(5))}),
|
||||
# stderr: DSA, USAGE, KV startup lines
|
||||
"usage_loaded": (re.compile(r"\[USAGE\].*?(\d+) selections"), lambda m: int(m.group(1))),
|
||||
"kv_slots": (re.compile(r"\[KV\].*?(\d+) context slots"), lambda m: int(m.group(1))),
|
||||
}
|
||||
|
||||
def extract_metrics(stdout: str, stderr: str) -> dict:
|
||||
"""Parse all metrics from the engine's stdout+stderr output."""
|
||||
text_out = stdout or ""
|
||||
text_err = stderr or ""
|
||||
metrics = {}
|
||||
# For most parsers we search BOTH streams (engine is inconsistent about which
|
||||
# channel a given line lands on). Some are stream-specific (noted below).
|
||||
combined = text_out + "\n" + text_err
|
||||
for name, (rx, fn) in RX.items():
|
||||
m = rx.search(combined)
|
||||
if m:
|
||||
try: metrics[name] = fn(m)
|
||||
except (ValueError, IndexError): pass
|
||||
# TOKENS dump is stderr-only and may appear once; grab it explicitly
|
||||
m = RX["tokens_dump"][0].search(text_err)
|
||||
if m:
|
||||
try: metrics["tokens_dump"] = RX["tokens_dump"][1](m)
|
||||
except: pass
|
||||
# Multiple CUDA devices — collect all
|
||||
cuda_devs = []
|
||||
for m in re.finditer(r"\[CUDA\] device (\d+): (.*?), ([\d.]+) GB VRAM, sm_(\d)(\d)", text_err):
|
||||
cuda_devs.append({"id":int(m.group(1)),"name":m.group(2).strip(),
|
||||
"vram_gb":float(m.group(3)),"sm":f"{m.group(4)}.{m.group(5)}"})
|
||||
if cuda_devs: metrics["cuda_devices"] = cuda_devs
|
||||
# Multiple progress checkpoints — collect the full curve
|
||||
progress = []
|
||||
for m in RX["progress_tps"][0].finditer(text_err):
|
||||
try:
|
||||
progress.append(RX["progress_tps"][1](m))
|
||||
except: pass
|
||||
if progress: metrics["progress_curve"] = progress
|
||||
# Multiple PROFILE lines — there are two (prefill + decode). Keep both.
|
||||
prof_lines = [(i, line) for i, line in enumerate(text_out.splitlines()) if line.startswith("PROFILE:")]
|
||||
for idx, (line_no, line) in enumerate(prof_lines):
|
||||
label = "prefill_profile" if idx == 0 else "decode_profile"
|
||||
p = {}
|
||||
for pname, (rx, fn) in RX.items():
|
||||
if not pname.startswith("prof_"): continue
|
||||
mm = rx.search(line)
|
||||
if mm:
|
||||
try: p[pname] = fn(mm)
|
||||
except: pass
|
||||
if p: metrics[label] = p
|
||||
return metrics
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ENGINE RUNNER — subprocess wrapper with timeout, signal handling, logging.
|
||||
# ---------------------------------------------------------------------------
|
||||
class EngineRunner:
|
||||
def __init__(self, glm_path, snap, out_dir, default_env=None, timeout=600):
|
||||
self.glm = str(glm_path)
|
||||
self.snap = str(snap)
|
||||
self.out_dir = Path(out_dir)
|
||||
self.out_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.default_env = default_env or {}
|
||||
self.timeout = timeout
|
||||
self.default_cap = 75 # production default (matches bench_full.sh)
|
||||
|
||||
def run_prompt(self, prompt, ngen=64, env_extra=None, log_name=None, cap=None):
|
||||
"""Run the engine in PROMPT mode and return (stdout, stderr, returncode, elapsed)."""
|
||||
env = dict(os.environ, SNAP=self.snap, PROMPT=prompt, NGEN=str(ngen))
|
||||
env.update(self.default_env)
|
||||
if env_extra: env.update(env_extra)
|
||||
env["TOKENS"] = "1" # always capture token ids for reliable text extraction
|
||||
cmd = [self.glm, str(cap if cap is not None else self.default_cap)]
|
||||
return self._exec(cmd, env, log_name)
|
||||
|
||||
def run_score(self, score_file, cap=None, env_extra=None, log_name=None):
|
||||
"""Run the engine in SCORE (log-likelihood) mode."""
|
||||
env = dict(os.environ, SNAP=self.snap, SCORE=score_file)
|
||||
env.update(self.default_env)
|
||||
if env_extra: env.update(env_extra)
|
||||
cmd = [self.glm, str(cap if cap is not None else self.default_cap)]
|
||||
return self._exec(cmd, env, log_name)
|
||||
|
||||
def _exec(self, cmd, env, log_name):
|
||||
t0 = time.time()
|
||||
log_path = self.out_dir / (log_name or f"run_{int(t0)}.log")
|
||||
try:
|
||||
proc = subprocess.Popen(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
try:
|
||||
stdout, stderr = proc.communicate(timeout=self.timeout)
|
||||
rc = proc.returncode
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
stdout, stderr = proc.communicate()
|
||||
rc = -1
|
||||
stderr = (stderr or "") + f"\n[TIMEOUT after {self.timeout}s]\n"
|
||||
except Exception as e:
|
||||
stdout, stderr, rc = "", f"[EXCEPTION] {e}\n{traceback.format_exc()}", -2
|
||||
elapsed = time.time() - t0
|
||||
# Write raw log (both streams, clearly delimited)
|
||||
with open(log_path, "w", encoding="utf-8") as f:
|
||||
f.write(f"=== CMD: {' '.join(cmd)}\n=== ELAPSED: {elapsed:.1f}s\n=== RC: {rc}\n\n")
|
||||
f.write("--- STDOUT ---\n"); f.write(stdout or ""); f.write("\n")
|
||||
f.write("--- STDERR ---\n"); f.write(stderr or ""); f.write("\n")
|
||||
return stdout or "", stderr or "", rc, elapsed
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# TEXT RECOVERY — decode the [TOKENS] ids back to text using the tokenizer.
|
||||
# Falls back to stdout extraction if tokenizer/tokens unavailable.
|
||||
# ---------------------------------------------------------------------------
|
||||
def recover_text(stdout, stderr, prompt, tokenizer=None):
|
||||
"""Extract generated text. Primary: TOKENS dump decoded. Fallback: stdout parse."""
|
||||
# Method 1: decode TOKENS ids via tokenizer
|
||||
if tokenizer:
|
||||
m = re.search(r"^\[TOKENS\] \d+ generated:(.*)$", stderr, re.MULTILINE)
|
||||
if m:
|
||||
ids = [int(x) for x in m.group(1).split()]
|
||||
if ids:
|
||||
try:
|
||||
return tokenizer.decode(ids), "tokens"
|
||||
except Exception: pass
|
||||
# Method 2: stdout — text is between the prompt string and "PROFILO" or "\n---"
|
||||
text = stdout or ""
|
||||
# Find the prompt in stdout, take everything after it up to PROFILO
|
||||
idx = text.find(prompt)
|
||||
if idx >= 0:
|
||||
after = text[idx + len(prompt):]
|
||||
cut = after.find("PROFILO")
|
||||
if cut >= 0:
|
||||
return after[:cut].strip(), "stdout"
|
||||
cut = after.find("\n---")
|
||||
if cut >= 0:
|
||||
return after[:cut].strip(), "stdout"
|
||||
return "", "none"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CORRUPTION / COHERENCE CHECKS
|
||||
# ---------------------------------------------------------------------------
|
||||
def check_repetition(token_ids):
|
||||
"""Detect degenerate repetition: same token 3+ consecutive times."""
|
||||
if not token_ids or len(token_ids) < 6:
|
||||
return False, 0
|
||||
max_run = 1; cur = 1
|
||||
for i in range(1, len(token_ids)):
|
||||
if token_ids[i] == token_ids[i-1]: cur += 1; max_run = max(max_run, cur)
|
||||
else: cur = 1
|
||||
return max_run >= 3, max_run
|
||||
|
||||
def check_expected(text, expect):
|
||||
"""Check if expected substring appears case-insensitively in the first 200 chars."""
|
||||
if not expect or not text: return None
|
||||
return expect.lower() in text[:200].lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PHASES
|
||||
# ---------------------------------------------------------------------------
|
||||
class DiagnosticHarness:
|
||||
def __init__(self, args):
|
||||
self.args = args
|
||||
self.snap = args.snap
|
||||
self.glm = args.glm
|
||||
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
self.out_dir = Path(args.out or f"./diag_results/{ts}")
|
||||
self.out_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Base env for all runs
|
||||
base_env = {"TEMP": "0", "PIPE": "1", "PIPE_WORKERS": "8", "DIRECT": "1"}
|
||||
if args.ram: base_env["RAM_GB"] = str(args.ram)
|
||||
if args.cuda:
|
||||
base_env.update({
|
||||
"COLI_CUDA": "1",
|
||||
"COLI_GPU": str(args.gpu) if args.gpu is not None else "0",
|
||||
"CUDA_DENSE": "1",
|
||||
"COLI_CUDA_ATTN": "1",
|
||||
"COLI_CUDA_PIPE": "2",
|
||||
"COLI_CUDA_PIPE_S_MIN": "1",
|
||||
})
|
||||
self.runner = EngineRunner(self.glm, self.snap, self.out_dir, base_env, timeout=args.timeout)
|
||||
self.runner.default_cap = args.cap
|
||||
# Load tokenizer for text recovery
|
||||
self.tokenizer = None
|
||||
tok_path = os.path.join(self.snap, "tokenizer.json")
|
||||
if os.path.exists(tok_path):
|
||||
try:
|
||||
from tokenizers import Tokenizer
|
||||
self.tokenizer = Tokenizer.from_file(tok_path)
|
||||
except Exception as e:
|
||||
print(f"[warn] could not load tokenizer: {e}", file=sys.stderr)
|
||||
self.results = {"meta": {"snap": self.snap, "glm": self.glm,
|
||||
"timestamp": ts, "args": vars(args)},
|
||||
"phases": {}}
|
||||
|
||||
def phase_system(self):
|
||||
"""Phase 0: minimal run to capture startup telemetry."""
|
||||
print("\n" + "="*60 + "\nPHASE 0: SYSTEM PROBE\n" + "="*60)
|
||||
stdout, stderr, rc, elapsed = self.runner.run_prompt(
|
||||
"Hello", ngen=1, log_name="system_probe.log")
|
||||
metrics = extract_metrics(stdout, stderr)
|
||||
result = {
|
||||
"rc": rc, "elapsed": elapsed,
|
||||
"load_secs": metrics.get("load_secs"),
|
||||
"resident_mb": metrics.get("resident_mb"),
|
||||
"n_layers": metrics.get("n_layers"),
|
||||
"n_experts": metrics.get("n_experts"),
|
||||
"mtp_status": metrics.get("mtp_status"),
|
||||
"idot_kernel": metrics.get("idot_kernel"),
|
||||
"cache_cap_requested": metrics.get("cache_cap"),
|
||||
"ram_budget_gb": metrics.get("ram_budget"),
|
||||
"cap_lowered": metrics.get("cap_lowered"),
|
||||
"cap_raised": metrics.get("cap_raised"),
|
||||
"cap_final": metrics.get("cap_ok"),
|
||||
"cuda_devices": metrics.get("cuda_devices", []),
|
||||
"cuda_mode": metrics.get("cuda_mode"),
|
||||
}
|
||||
# Print summary
|
||||
print(f" load time: {result['load_secs']:.2f}s" if result['load_secs'] else " load time: FAILED")
|
||||
print(f" resident: {result['resident_mb']:.0f} MB" if result['resident_mb'] else "")
|
||||
print(f" layers/exp: {result['n_layers']}/{result['n_experts']}" if result['n_layers'] else "")
|
||||
print(f" MTP: {result['mtp_status']}")
|
||||
print(f" idot kernel: {result['idot_kernel']}")
|
||||
print(f" RAM budget: {result['ram_budget_gb']} GB" if result['ram_budget_gb'] else " RAM budget: auto")
|
||||
if result['cap_lowered']:
|
||||
print(f" cache cap: {result['cap_lowered'][0]} -> {result['cap_lowered'][1]} (RAM-lowered)")
|
||||
elif result['cap_final']:
|
||||
print(f" cache cap: {result['cap_final']} (ok)")
|
||||
else:
|
||||
print(f" cache cap: {result['cache_cap_requested']}")
|
||||
for d in result['cuda_devices']:
|
||||
print(f" GPU {d['id']}: {d['name']}, {d['vram_gb']:.1f} GB, sm_{d['sm']}")
|
||||
if rc != 0 and rc != -1:
|
||||
print(f" WARNING: engine returned rc={rc}")
|
||||
self.results["phases"]["system"] = result
|
||||
return result
|
||||
|
||||
def phase_smoke(self):
|
||||
"""Phase 1: correctness smoke test across curated prompts."""
|
||||
print("\n" + "="*60 + "\nPHASE 1: CORRECTNESS SMOKE TEST\n" + "="*60)
|
||||
ngen = self.args.ngen
|
||||
prompt_results = []
|
||||
pass_count = 0; total = len(PROMPTS)
|
||||
for p in PROMPTS:
|
||||
stdout, stderr, rc, elapsed = self.runner.run_prompt(
|
||||
p["prompt"], ngen=ngen, log_name=f"smoke_{p['id']}.log")
|
||||
metrics = extract_metrics(stdout, stderr)
|
||||
token_ids = metrics.get("tokens_dump", [])
|
||||
text, method = recover_text(stdout, stderr, p["prompt"], self.tokenizer)
|
||||
is_rep, max_run = check_repetition(token_ids)
|
||||
expect_ok = check_expected(text, p.get("expect"))
|
||||
# Verdict: pass if text is non-empty, no severe repetition, and (if factual) expected found
|
||||
has_text = bool(text.strip())
|
||||
ok = has_text and not is_rep
|
||||
if p.get("expect") and expect_ok is False: ok = False
|
||||
if ok: pass_count += 1
|
||||
# Diagnose failure mode for reporting
|
||||
if not has_text:
|
||||
fail_reason = "no tokens generated (immediate EOS)"
|
||||
elif is_rep:
|
||||
fail_reason = f"repetition loop (max run={max_run})"
|
||||
elif p.get("expect") and expect_ok is False:
|
||||
fail_reason = f"expected '{p['expect']}' not found"
|
||||
else:
|
||||
fail_reason = ""
|
||||
prompt_results.append({
|
||||
"id": p["id"], "cat": p["cat"], "prompt": p["prompt"],
|
||||
"expect": p.get("expect"), "generated": text[:300],
|
||||
"expect_match": expect_ok, "repetition": is_rep, "max_run": max_run,
|
||||
"toks_generated": len(token_ids), "decode_tps": metrics.get("decode_tps"),
|
||||
"hit_rate": metrics.get("hit_rate"), "rc": rc, "elapsed": elapsed,
|
||||
"text_method": method, "pass": ok, "fail_reason": fail_reason,
|
||||
})
|
||||
status = "PASS" if ok else "FAIL"
|
||||
extra = f" expect={'Y' if expect_ok else ('N' if expect_ok is False else '-')}" if p.get("expect") else ""
|
||||
tps = f" {metrics.get('decode_tps',0):.2f}t/s" if metrics.get("decode_tps") else ""
|
||||
print(f" [{status}] {p['id']:<22} ({p['cat']:<11}) rep={'Y' if is_rep else 'N'}{extra}{tps}")
|
||||
if text:
|
||||
preview = text.replace("\n", " ")[:80]
|
||||
print(f" -> {preview}")
|
||||
elif rc != 0:
|
||||
print(f" -> [engine rc={rc}]")
|
||||
else:
|
||||
print(f" -> [{fail_reason}]")
|
||||
result = {"prompts": prompt_results, "pass": pass_count, "total": total,
|
||||
"pass_rate": 100.0 * pass_count / total if total else 0}
|
||||
print(f"\n SMOKE SUMMARY: {pass_count}/{total} passed ({result['pass_rate']:.0f}%)")
|
||||
self.results["phases"]["smoke"] = result
|
||||
return result
|
||||
|
||||
def phase_diagnostic(self):
|
||||
"""Phase 2: single deep-instrumented run — full PROFILE + routing + MTP."""
|
||||
print("\n" + "="*60 + "\nPHASE 2: FULL SYSTEM DIAGNOSTIC\n" + "="*60)
|
||||
diag_env = {
|
||||
"LOOKA": "1", "DISK_SPLIT": "1", "ROUTE_AGREE": "1",
|
||||
}
|
||||
if self.args.cuda:
|
||||
diag_env["COLI_CUDA_PROFILE"] = "1"
|
||||
prompt = ("Write a short paragraph explaining how photosynthesis works. "
|
||||
"Include the roles of sunlight, water, and carbon dioxide.")
|
||||
stdout, stderr, rc, elapsed = self.runner.run_prompt(
|
||||
prompt, ngen=self.args.ngen, env_extra=diag_env, log_name="diagnostic.log")
|
||||
metrics = extract_metrics(stdout, stderr)
|
||||
text, _ = recover_text(stdout, stderr, prompt, self.tokenizer)
|
||||
result = {
|
||||
"rc": rc, "elapsed": elapsed,
|
||||
"generated_text": text[:500],
|
||||
"prefill_profile": metrics.get("prefill_profile", {}),
|
||||
"decode_profile": metrics.get("decode_profile", {}),
|
||||
"decode_tps": metrics.get("decode_tps"),
|
||||
"prefill_secs": metrics.get("prefill_secs"),
|
||||
"hit_rate": metrics.get("hit_rate"),
|
||||
"rss_gb": metrics.get("rss_gb"),
|
||||
"experts_per_tok": metrics.get("experts_per_tok"),
|
||||
"mtp_accept": metrics.get("mtp_accept"),
|
||||
"mtp_counts": metrics.get("mtp_acc_cnt"),
|
||||
"spec_tok_per_fw": metrics.get("spec_tok_per_fw"),
|
||||
"cuda_tier": metrics.get("cuda_tier"),
|
||||
"progress_curve": metrics.get("progress_curve", []),
|
||||
}
|
||||
# Print the PROFILE breakdown
|
||||
dp = result["decode_profile"]
|
||||
print(f"\n DECODE TIMING BREAKDOWN (per-bucket, seconds):")
|
||||
print(f" expert-disk: {dp.get('prof_expert_disk', '?'):>8}")
|
||||
print(f" expert-matmul: {dp.get('prof_expert_mm', '?'):>8}")
|
||||
print(f" attention: {dp.get('prof_attention', '?'):>8}")
|
||||
print(f" lm_head: {dp.get('prof_lm_head', '?'):>8}")
|
||||
print(f" other: {dp.get('prof_other', '?'):>8}")
|
||||
print(f"\n PERFORMANCE:")
|
||||
print(f" prefill: {result['prefill_secs']:.2f}s" if result['prefill_secs'] else " prefill: ?")
|
||||
print(f" decode: {result['decode_tps']:.3f} tok/s" if result['decode_tps'] else " decode: ?")
|
||||
print(f" hit rate: {result['hit_rate']:.1f}%" if result.get('hit_rate') is not None else " hit rate: ?")
|
||||
print(f" RSS: {result['rss_gb']:.1f} GB" if result.get('rss_gb') else " RSS: ?")
|
||||
print(f" exp/tok: {result['experts_per_tok']:.1f}" if result.get('experts_per_tok') else " exp/tok: ?")
|
||||
print(f" MTP: {result['mtp_accept']:.0f}% accept" if result.get('mtp_accept') is not None else " MTP: ?")
|
||||
if result.get('cuda_tier'):
|
||||
ct = result['cuda_tier']
|
||||
print(f" CUDA tier: {ct['resident']} experts, {ct['vram_gb']:.1f} GB VRAM")
|
||||
# Show generated text preview
|
||||
if text:
|
||||
print(f"\n GENERATED TEXT (first 200 chars):")
|
||||
print(f" {text[:200].replace(chr(10), ' ')}")
|
||||
else:
|
||||
print(f"\n GENERATED TEXT: [none recovered]")
|
||||
self.results["phases"]["diagnostic"] = result
|
||||
return result
|
||||
|
||||
def phase_quality(self):
|
||||
"""Phase 3: benchmark accuracy via eval_glm.py SCORE mode."""
|
||||
print("\n" + "="*60 + "\nPHASE 3: QUALITY BENCHMARKS\n" + "="*60)
|
||||
eval_script = os.path.join(os.path.dirname(__file__), "eval_glm.py")
|
||||
bench_dir = os.path.join(os.path.dirname(os.path.dirname(__file__)), "bench")
|
||||
if not os.path.exists(eval_script):
|
||||
print(f" [SKIP] eval_glm.py not found at {eval_script}")
|
||||
self.results["phases"]["quality"] = {"error": "eval_glm.py not found"}
|
||||
return None
|
||||
tasks = ["hellaswag", "arc_challenge", "mmlu"]
|
||||
missing = [t for t in tasks if not os.path.exists(os.path.join(bench_dir, f"{t}.jsonl"))]
|
||||
if missing:
|
||||
print(f" [SKIP] benchmark data missing: {missing}")
|
||||
print(f" run: python tools/fetch_benchmarks.py --out {bench_dir}")
|
||||
self.results["phases"]["quality"] = {"error": f"missing benchmark data: {missing}"}
|
||||
return None
|
||||
limit = self.args.quality_limit
|
||||
py = sys.executable
|
||||
cmd = [py, eval_script, "--snap", self.snap, "--glm", self.glm,
|
||||
"--data", bench_dir, "--tasks", ",".join(tasks), "--limit", str(limit)]
|
||||
env = dict(os.environ)
|
||||
if self.args.ram: env["RAM_GB"] = str(self.args.ram)
|
||||
print(f" Running eval_glm.py (tasks={tasks}, n={limit})...")
|
||||
print(f" This takes ~{limit*3*4/0.05:.0f}s at 0.05 tok/s (worst case)...")
|
||||
t0 = time.time()
|
||||
log_path = self.out_dir / "quality_eval.log"
|
||||
try:
|
||||
with open(log_path, "w", encoding="utf-8") as logf:
|
||||
proc = subprocess.run(cmd, env=env, capture_output=True, text=True, timeout=self.args.timeout*3,
|
||||
encoding="utf-8", errors="replace")
|
||||
logf.write(proc.stdout); logf.write("\n--- STDERR ---\n"); logf.write(proc.stderr)
|
||||
elapsed = time.time() - t0
|
||||
# Parse the acc/acc_norm table from eval_glm.py output
|
||||
scores = {}
|
||||
for line in proc.stdout.splitlines():
|
||||
# lines like: "hellaswag 40 45.0% 50.0%"
|
||||
m = re.match(r"(\w+)\s+(\d+)\s+([\d.]+)%\s+([\d.]+)%", line.strip())
|
||||
if m:
|
||||
scores[m.group(1)] = {"n": int(m.group(2)), "acc": float(m.group(3)),
|
||||
"acc_norm": float(m.group(4))}
|
||||
mean_m = re.search(r"MEAN acc_norm:\s*([\d.]+)%", proc.stdout)
|
||||
result = {"scores": scores, "mean_acc_norm": float(mean_m.group(1)) if mean_m else None,
|
||||
"elapsed": elapsed, "rc": proc.returncode}
|
||||
print(f" Completed in {elapsed:.0f}s\n")
|
||||
print(f" {'task':<18} {'n':>4} {'acc':>7} {'acc_norm':>9}")
|
||||
for t, s in scores.items():
|
||||
print(f" {t:<18} {s['n']:>4} {s['acc']:>6.1f}% {s['acc_norm']:>8.1f}%")
|
||||
if result["mean_acc_norm"] is not None:
|
||||
print(f"\n MEAN acc_norm: {result['mean_acc_norm']:.1f}%")
|
||||
except subprocess.TimeoutExpired:
|
||||
result = {"error": f"eval timed out after {self.args.timeout*3}s"}
|
||||
print(f" [TIMEOUT] eval_glm.py exceeded {self.args.timeout*3}s")
|
||||
except Exception as e:
|
||||
result = {"error": str(e)}
|
||||
print(f" [ERROR] {e}")
|
||||
self.results["phases"]["quality"] = result
|
||||
return result
|
||||
|
||||
def phase_throughput(self):
|
||||
"""Phase 4: tok/s comparison — MTP on vs off."""
|
||||
print("\n" + "="*60 + "\nPHASE 4: THROUGHPUT BENCHMARK\n" + "="*60)
|
||||
prompt = "Summarize the plot of Romeo and Juliet in three sentences."
|
||||
ngen = self.args.ngen
|
||||
results = {}
|
||||
# Run 1: with MTP (default)
|
||||
print(f" [1/2] MTP ON (draft=3)...")
|
||||
stdout, stderr, rc, t = self.runner.run_prompt(
|
||||
prompt, ngen=ngen, log_name="throughput_mtp_on.log")
|
||||
m_on = extract_metrics(stdout, stderr)
|
||||
results["mtp_on"] = {"tps": m_on.get("decode_tps"), "hit": m_on.get("hit_rate"),
|
||||
"mtp_accept": m_on.get("mtp_accept"), "elapsed": t}
|
||||
print(f" {m_on.get('decode_tps',0):.3f} tok/s | hit {m_on.get('hit_rate',0):.1f}% | "
|
||||
f"MTP {m_on.get('mtp_accept',0):.0f}%")
|
||||
# Run 2: MTP off
|
||||
print(f" [2/2] MTP OFF (MTP=0)...")
|
||||
stdout, stderr, rc, t = self.runner.run_prompt(
|
||||
prompt, ngen=ngen, env_extra={"MTP": "0"}, log_name="throughput_mtp_off.log")
|
||||
m_off = extract_metrics(stdout, stderr)
|
||||
results["mtp_off"] = {"tps": m_off.get("decode_tps"), "hit": m_off.get("hit_rate"),
|
||||
"elapsed": t}
|
||||
print(f" {m_off.get('decode_tps',0):.3f} tok/s | hit {m_off.get('hit_rate',0):.1f}%")
|
||||
# Compute speedup
|
||||
if results["mtp_on"]["tps"] and results["mtp_off"]["tps"] and results["mtp_off"]["tps"] > 0:
|
||||
sp = results["mtp_on"]["tps"] / results["mtp_off"]["tps"]
|
||||
results["mtp_speedup"] = sp
|
||||
print(f"\n MTP speedup: {sp:.2f}x ({'MTP helps' if sp > 1.05 else 'MTP hurts' if sp < 0.95 else 'no effect'})")
|
||||
else:
|
||||
print(f"\n MTP speedup: (insufficient data)")
|
||||
self.results["phases"]["throughput"] = results
|
||||
return results
|
||||
|
||||
def write_report(self):
|
||||
"""Write report.json and report.md."""
|
||||
ts = self.results["meta"]["timestamp"]
|
||||
json_path = self.out_dir / "report.json"
|
||||
md_path = self.out_dir / "report.md"
|
||||
# JSON
|
||||
with open(json_path, "w", encoding="utf-8") as f:
|
||||
json.dump(self.results, f, indent=2, default=str)
|
||||
# Markdown
|
||||
lines = [
|
||||
f"# Diagnostic Report — {ts}",
|
||||
f"",
|
||||
f"**Model:** `{self.snap}`",
|
||||
f"**Engine:** `{self.glm}`",
|
||||
f"**Date:** {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}",
|
||||
f"",
|
||||
]
|
||||
phases = self.results["phases"]
|
||||
# System
|
||||
if "system" in phases:
|
||||
s = phases["system"]
|
||||
lines += ["## Phase 0: System Probe", ""]
|
||||
lines += [f"- Load time: **{s.get('load_secs','?')}s**"]
|
||||
lines += [f"- Layers/experts: {s.get('n_layers','?')}/{s.get('n_experts','?')}"]
|
||||
lines += [f"- MTP: {s.get('mtp_status','?')}"]
|
||||
lines += [f"- idot kernel: `{s.get('idot_kernel','?')}`"]
|
||||
lines += [f"- RAM budget: {s.get('ram_budget_gb','auto')} GB"]
|
||||
if s.get("cap_lowered"):
|
||||
lines += [f"- Cache cap: {s['cap_lowered'][0]}→{s['cap_lowered'][1]} (RAM-lowered)"]
|
||||
elif s.get("cap_final"):
|
||||
lines += [f"- Cache cap: {s['cap_final']}"]
|
||||
for d in s.get("cuda_devices", []):
|
||||
lines += [f"- GPU {d['id']}: {d['name']}, {d['vram_gb']:.1f} GB, sm_{d['sm']}"]
|
||||
lines.append("")
|
||||
# Smoke
|
||||
if "smoke" in phases:
|
||||
sm = phases["smoke"]
|
||||
lines += [f"## Phase 1: Correctness Smoke", "",
|
||||
f"**{sm['pass']}/{sm['total']} prompts passed ({sm['pass_rate']:.0f}%)**", "",
|
||||
"| ID | Category | Pass | Expect | Repetition | tok/s | Generated (first 80 chars) |",
|
||||
"|---|---|---|---|---|---|---|"]
|
||||
for p in sm["prompts"]:
|
||||
gen = p["generated"][:80].replace("|", "\\|").replace("\n", " ") if p["generated"] else ""
|
||||
exp = "Y" if p.get("expect_match") else ("N" if p.get("expect_match") is False else "-")
|
||||
tps = f"{p.get('decode_tps',0):.2f}" if p.get("decode_tps") else "-"
|
||||
lines.append(f"| {p['id']} | {p['cat']} | {'✅' if p['pass'] else '❌'} | {exp} | "
|
||||
f"{'⚠️' if p['repetition'] else '-'} (run={p.get('max_run',0)}) | {tps} | {gen} |")
|
||||
lines.append("")
|
||||
# Diagnostic
|
||||
if "diagnostic" in phases:
|
||||
d = phases["diagnostic"]
|
||||
lines += ["## Phase 2: Full System Diagnostic", ""]
|
||||
dp = d.get("decode_profile", {})
|
||||
lines += ["### Decode Timing Breakdown", "",
|
||||
"| Bucket | Seconds |", "|---|---|"]
|
||||
for k, label in [("prof_expert_disk","expert-disk"),("prof_expert_mm","expert-matmul"),
|
||||
("prof_attention","attention"),("prof_lm_head","lm_head"),
|
||||
("prof_other","other")]:
|
||||
v = dp.get(k, "?")
|
||||
lines.append(f"| {label} | {v} |")
|
||||
lines += [f"", f"### Performance", f"- Decode: **{d.get('decode_tps','?')} tok/s**",
|
||||
f"- Prefill: {d.get('prefill_secs','?')}s",
|
||||
f"- Expert hit rate: {d.get('hit_rate','?')}%",
|
||||
f"- RSS: {d.get('rss_gb','?')} GB",
|
||||
f"- Experts/token: {d.get('experts_per_tok','?')}",
|
||||
f"- MTP acceptance: {d.get('mtp_accept','?')}%", ""]
|
||||
if d.get("generated_text"):
|
||||
lines += [f"### Generated Text", f"```\n{d['generated_text'][:300]}\n```", ""]
|
||||
# Quality
|
||||
if "quality" in phases:
|
||||
q = phases["quality"]
|
||||
lines += ["## Phase 3: Quality Benchmarks", ""]
|
||||
if "error" in q:
|
||||
lines += [f"⚠️ {q['error']}", ""]
|
||||
else:
|
||||
lines += [f"**Mean acc_norm: {q.get('mean_acc_norm','?')}%**", "",
|
||||
"| Task | n | acc | acc_norm |", "|---|---|---|---|"]
|
||||
for t, s in q.get("scores", {}).items():
|
||||
lines.append(f"| {t} | {s['n']} | {s['acc']:.1f}% | {s['acc_norm']:.1f}% |")
|
||||
lines.append("")
|
||||
# Throughput
|
||||
if "throughput" in phases:
|
||||
th = phases["throughput"]
|
||||
lines += ["## Phase 4: Throughput", "",
|
||||
"| Mode | tok/s | hit% | MTP accept |", "|---|---|---|---|"]
|
||||
on = th.get("mtp_on", {}); off = th.get("mtp_off", {})
|
||||
lines.append(f"| MTP ON | {on.get('tps','?')} | {on.get('hit','?')}% | {on.get('mtp_accept','?')}% |")
|
||||
lines.append(f"| MTP OFF | {off.get('tps','?')} | {off.get('hit','?')}% | — |")
|
||||
if th.get("mtp_speedup"):
|
||||
lines.append(f"\n**MTP speedup: {th['mtp_speedup']:.2f}x**")
|
||||
lines.append("")
|
||||
with open(md_path, "w", encoding="utf-8") as f:
|
||||
f.write("\n".join(lines))
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Report written:")
|
||||
print(f" JSON: {json_path}")
|
||||
print(f" MD: {md_path}")
|
||||
print(f" Logs: {self.out_dir}/*.log")
|
||||
print(f"{'='*60}")
|
||||
|
||||
def run(self):
|
||||
phase = self.args.phase
|
||||
if phase == "all":
|
||||
self.phase_system()
|
||||
self.phase_smoke()
|
||||
self.phase_diagnostic()
|
||||
self.phase_quality()
|
||||
self.phase_throughput()
|
||||
elif phase == "system": self.phase_system()
|
||||
elif phase == "smoke": self.phase_smoke()
|
||||
elif phase == "diagnostic": self.phase_diagnostic()
|
||||
elif phase == "quality": self.phase_quality()
|
||||
elif phase == "throughput": self.phase_throughput()
|
||||
else:
|
||||
print(f"Unknown phase: {phase}", file=sys.stderr); sys.exit(1)
|
||||
self.write_report()
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Comprehensive model diagnostic harness for colibri GLM-5.2")
|
||||
ap.add_argument("--snap", required=True, help="model snapshot directory")
|
||||
ap.add_argument("--glm", default=None, help="engine binary path (default: ./glm.exe or ./glm)")
|
||||
ap.add_argument("--phase", default="all",
|
||||
choices=["all","system","smoke","diagnostic","quality","throughput"],
|
||||
help="which test phase to run")
|
||||
ap.add_argument("--out", default=None, help="output directory (default: ./diag_results/<timestamp>)")
|
||||
ap.add_argument("--ngen", type=int, default=64, help="generation length for smoke/throughput (default 64)")
|
||||
ap.add_argument("--quality-limit", type=int, default=40, help="questions per benchmark task (default 40)")
|
||||
ap.add_argument("--ram", type=float, default=0, help="RAM_GB override (0=auto)")
|
||||
ap.add_argument("--cuda", action="store_true", help="enable COLI_CUDA GPU tier")
|
||||
ap.add_argument("--gpu", type=int, default=None, help="GPU device ordinal (with --cuda)")
|
||||
ap.add_argument("--cap", type=int, default=75, help="experts-per-layer cache cap (default 75)")
|
||||
ap.add_argument("--timeout", type=int, default=600, help="per-run timeout in seconds (default 600)")
|
||||
a = ap.parse_args()
|
||||
# Auto-detect engine binary
|
||||
if a.glm is None:
|
||||
for cand in ["./glm.exe", "./glm", "../glm.exe", os.path.join(os.path.dirname(__file__), "..", "glm.exe")]:
|
||||
if os.path.exists(cand): a.glm = os.path.abspath(cand); break
|
||||
if a.glm is None:
|
||||
print("ERROR: could not find glm/glm.exe. Specify with --glm", file=sys.stderr); sys.exit(1)
|
||||
if not os.path.isdir(a.snap):
|
||||
print(f"ERROR: snapshot dir does not exist: {a.snap}", file=sys.stderr); sys.exit(1)
|
||||
harness = DiagnosticHarness(a)
|
||||
harness.run()
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,458 @@
|
||||
"""Efficiency / regression harness for the colibri engine.
|
||||
|
||||
The engine already emits rich telemetry (REPLAY tok/s, PROFILE phase timings,
|
||||
[PROF] time shares + verdict, CUDA expert-tier utilization). Until now every
|
||||
consumer of that telemetry — `benchmark_cuda_fixture.py`, `bench_full.sh`,
|
||||
`bench_ux.sh` — has only *printed* it for a human to eyeball. This module turns
|
||||
each signal into a parseable field so tests can assert on it.
|
||||
|
||||
Design:
|
||||
- Reuses SPEED_RE / PROFILE_RE from tools.benchmark_cuda_fixture (no drift).
|
||||
- parse_run() is pure: stdout+stderr in, dict out. Easy to unit-test against
|
||||
captured strings (like the existing test_benchmark_cuda_fixture does).
|
||||
- run_engine() is the subprocess wrapper. Captures stdout and stderr
|
||||
separately, because the engine splits them: PROFILE/REPLAY/CUDA-tier go to
|
||||
stdout, the [CUDA]/[PROF]/[prefill] banners go to stderr.
|
||||
- Floor defaults are module constants (tunable in one place, not scattered).
|
||||
|
||||
No model file is required to import this module; only run_engine() invokes the
|
||||
binary. parse_run() works on any captured text, so most test surface is covered
|
||||
by string fixtures without spinning the engine at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
# Reuse the validated regexes from the existing A/B benchmark harness so the
|
||||
# PROFILE field order (disk, expert_matmul, attention, lm_head, other) and the
|
||||
# tok/s capture stay identical. Drift here would silently break every consumer.
|
||||
from tools.benchmark_cuda_fixture import SPEED_RE as _SPEED_RE_REPLAY, PROFILE_RE, PROFILE_KEYS
|
||||
|
||||
# SPEED_RE (from benchmark_cuda_fixture) matches the REPLAY-mode line only:
|
||||
# "REPLAY decode: ... | 12.34 tok/s | ..."
|
||||
# run_text / PROMPT mode uses a DIFFERENT format (glm.c:4682):
|
||||
# "decode N tokens in X.XXs (12.34 tok/s) | expert hit rate ..."
|
||||
# This alt regex catches the parenthesized form so the full-model report (which
|
||||
# uses PROMPT mode) gets a real tok/s instead of reporting it missing.
|
||||
SPEED_RE_TEXT = re.compile(r"decode \d+ tokens in [0-9.]+s \(([0-9.]+) tok/s\)")
|
||||
|
||||
|
||||
def _first_speed(stdout: str):
|
||||
"""Find tok/s in whichever run-mode format the engine used."""
|
||||
for rx in (_SPEED_RE_REPLAY, SPEED_RE_TEXT):
|
||||
m = rx.search(stdout)
|
||||
if m:
|
||||
return m
|
||||
return None
|
||||
|
||||
|
||||
# Public alias so existing imports keep working (tests reference SPEED_RE).
|
||||
SPEED_RE = _SPEED_RE_REPLAY
|
||||
|
||||
|
||||
# --- additional parsers (formats verified against glm.c printf strings) ---
|
||||
|
||||
# "expert hit rate 88.1%" (summary line) | "expert hit 95.0%" (REPLAY line)
|
||||
HIT_RE = re.compile(r"expert hit(?:\s+rate)?\s+([0-9.]+)%")
|
||||
|
||||
# "[PROF] time shares: expert-I/O 3% | expert-matmul 12% | attention 56% | lm_head 2% | other 27%"
|
||||
SHARES_RE = re.compile(
|
||||
r"\[PROF\] time shares: expert-I/O\s+([0-9.]+)%\s*\|\s*expert-matmul\s+([0-9.]+)%\s*"
|
||||
r"\|\s*attention\s+([0-9.]+)%\s*\|\s*lm_head\s+([0-9.]+)%\s*\|\s*other\s+([0-9.]+)%"
|
||||
)
|
||||
|
||||
# "[PROF] verdict: I/O-bound — 60% of the time ..." (also compute-bound / attention-bound / balanced)
|
||||
VERDICT_RE = re.compile(r"\[PROF\] verdict:\s*(I/O-bound|compute-bound|attention-bound|balanced)")
|
||||
|
||||
# "[PROF] expert I/O: ... hit 95.0% (76 hit / 4 load) | 4.0 loads/token"
|
||||
LOADS_PER_TOK_RE = re.compile(r"\|\s*([0-9.]+)\s+loads/token")
|
||||
|
||||
# "CUDA expert tier: 111 resident experts (2.36 GB) | 5400 calls served from VRAM" (stdout)
|
||||
CUDA_TIER_RE = re.compile(
|
||||
r"CUDA expert tier:\s+(\d+)\s+resident experts\s+\(([0-9.]+)\s+GB\)\s*\|\s+(\d+)\s+calls served from VRAM"
|
||||
)
|
||||
|
||||
# "[CUDA] device 0: NVIDIA ..., 14.4 GB VRAM, sm_120" (stderr, per device at init)
|
||||
CUDA_DEVICE_RE = re.compile(r"\[CUDA\] device\s+\d+:")
|
||||
|
||||
# "[CUDA] resident set: 12 tensors, 0.45 GB VRAM" (stderr, cuda_stats_print)
|
||||
CUDA_RESIDENT_RE = re.compile(
|
||||
r"\[CUDA\] resident set:\s+(\d+)\s+tensors,\s+([0-9.]+)\s+GB\s+VRAM"
|
||||
)
|
||||
|
||||
# "PREFILL (teacher-forcing) C vs oracle: 11/32 positions | 1700.4 pos/s" (TF=1 mode, stdout)
|
||||
TF_MATCH_RE = re.compile(r"PREFILL \(teacher-forcing\).*:\s+(\d+)/(\d+)\s+positions")
|
||||
|
||||
# --- the six deeper signals (added after "are you gathering everything?" audit) ---
|
||||
|
||||
# "ATTENTION: projection/RoPE 0.050s | score-softmax-value 0.009s | output projection 0.011s"
|
||||
# Sub-breakdown of the attention phase — answers "how is attention being read".
|
||||
ATTN_BREAKDOWN_RE = re.compile(
|
||||
r"ATTENTION: projection/RoPE\s+([0-9.]+)s\s*\|\s*score-softmax-value\s+([0-9.]+)s\s*"
|
||||
r"\|\s*output projection\s+([0-9.]+)s"
|
||||
)
|
||||
|
||||
# "[PROF] decode forwards: 20 | latency p50 7.3 ms | p90 8.0 ms | p99 8.1 ms | max 8.2 ms | 1.00 tok/forward"
|
||||
# Per-forward tail latency — a p99 >> p50 means decode stalls (I/O hiccups, KV grow).
|
||||
LATENCY_RE = re.compile(
|
||||
r"\[PROF\] decode forwards:\s+(\d+)\s*\| latency p50\s+([0-9.]+)\s*ms\s*\| "
|
||||
r"p90\s+([0-9.]+)\s*ms\s*\|\s*p99\s+([0-9.]+)\s*ms\s*\|\s*max\s+([0-9.]+)\s*ms"
|
||||
)
|
||||
|
||||
# "[PROF] expert I/O: 0.004 GB fetched (0.2 MB/token, 0.03 GB/s over the run) | hit 95.0% ... |
|
||||
# 4.0 loads/token | 0.0s read service / 0.0s felt wait"
|
||||
# Absolute disk throughput + the felt-wait split (the [PROF] version, more detailed than PROFILE).
|
||||
EXPERT_IO_RE = re.compile(
|
||||
r"\[PROF\] expert I/O:\s+([0-9.]+)\s+GB fetched\s+\(([0-9.]+)\s+MB/token,\s*([0-9.]+)\s+GB/s"
|
||||
r".*?\|\s*([0-9.]+)\s+loads/token\s*\|\s*([0-9.]+)s\s+read service\s*/\s*([0-9.]+)s\s+felt wait"
|
||||
)
|
||||
|
||||
# "speculation: 1.05 tokens/forward (19 forwards per 20 tokens) | MTP acceptance 44% (7/16)"
|
||||
# Draft efficiency — is the speculative decoder pulling weight or dead overhead?
|
||||
SPECULATION_RE = re.compile(
|
||||
r"speculation:\s+([0-9.]+)\s+tokens/forward\s+\((\d+)\s+forwards per\s+(\d+)\s+tokens\)"
|
||||
r"\s*\|\s*MTP acceptance\s+([0-9.]+)%"
|
||||
)
|
||||
|
||||
# "experts loaded/token: 450.0 (per-layer 56.25 across 8; baseline topk=8) | TOPK=0 TOPP=0.00"
|
||||
# Fuller than loads_per_tok: includes the per-layer spread + the active topk/topp.
|
||||
EXPERTS_LOADED_RE = re.compile(
|
||||
r"experts loaded/token:\s+([0-9.]+)\s+\(per-layer\s+([0-9.]+)\s+across\s+(\d+);\s*baseline topk=(\d+)\)"
|
||||
)
|
||||
|
||||
# "[PROF] machine: Intel(...) | 22 cores (22 omp threads) | RAM 34.1 GB total, 27.1 GB available | backend CUDA"
|
||||
# Provenance — makes a report reproducible across machines/runs.
|
||||
MACHINE_RE = re.compile(r"\[PROF\] machine:\s*(.*)\|\s*(\d+)\s+cores.*backend\s+(\S+)")
|
||||
|
||||
# "[PROF] config: RAM_GB=auto 23.9 CTX=4096 | expert cache cap 8/layer ... | DRAFT=0 PIPE=1 DIRECT=0 ..."
|
||||
# Effective resolved config (after auto-budgeting). Answers "what config actually ran".
|
||||
CONFIG_RE = re.compile(r"\[PROF\] config:\s*(.*)")
|
||||
|
||||
# "[ORACLE] mismatch pos=7 expected=197 got=22" (TF=1 mode, stderr, per position)
|
||||
# Captures the engine's *actual* argmax at each position, so two backends can be
|
||||
# compared DIRECTLY (independent of how each relates to the oracle).
|
||||
TF_MISMATCH_RE = re.compile(r"\[ORACLE\] mismatch pos=(\d+) expected=\d+ got=(\d+)")
|
||||
|
||||
# --- routing-quality + disk-split + cuda-groups (the deep signals) ---
|
||||
|
||||
# summary suffix: " | swap 3.1% (12/384)" (CACHE_ROUTE inline)
|
||||
SWAP_RE = re.compile(r"swap\s+([0-9.]+)%\s+\((\d+)/(\d+)\)")
|
||||
|
||||
# summary suffix: " | route_agree 85.0% | route_kl 0.0123" (CACHE_ROUTE/ROUTE_AGREE)
|
||||
ROUTE_AGREE_RE = re.compile(r"route_agree\s+([0-9.]+)%\s*\|\s*route_kl\s+([0-9.]+)")
|
||||
|
||||
# "disk-load split: draft 8 + absorb 0 + verify/main 3 misses | MTP-layer 0 loads 0.00 GB |
|
||||
# main-layers 11 loads 0.01 GB (MTP 0.0% of bytes)" (DISK_SPLIT=1)
|
||||
DISK_SPLIT_RE = re.compile(
|
||||
r"disk-load split: draft\s+(\d+)\s+\+\s+absorb\s+(\d+)\s+\+\s+verify/main\s+(\d+)\s+misses"
|
||||
r"\s*\|\s*MTP-layer\s+(\d+)\s+loads\s+([0-9.]+)\s+GB\s*\|\s*main-layers\s+(\d+)\s+loads\s+([0-9.]+)\s+GB"
|
||||
r"(?:\s+\(MTP\s+([0-9.]+)%\s+of bytes\))?"
|
||||
)
|
||||
|
||||
# "[CUDA] expert groups: 120 call, 840 expert, 1200 righe (7.00 expert/call)"
|
||||
CUDA_GROUPS_RE = re.compile(
|
||||
r"\[CUDA\] expert groups:\s+(\d+)\s+call,\s+(\d+)\s+expert,\s+(\d+)\s+righe\s+\(([0-9.]+)\s+expert/call\)"
|
||||
)
|
||||
|
||||
# "[CUDA] expert groups timing: H2D 12.3 ms | kernel 45.6 ms | D2H 7.8 ms" (COLI_CUDA_PROFILE=1)
|
||||
CUDA_GROUPS_TIME_RE = re.compile(
|
||||
r"\[CUDA\] expert groups timing: H2D\s+([0-9.]+)\s+ms\s*\|\s*kernel\s+([0-9.]+)\s+ms\s*\|\s*D2H\s+([0-9.]+)\s+ms"
|
||||
)
|
||||
|
||||
# LOOKAHEAD recall block — 4 named rows. (name, pct, hit, tot)
|
||||
LOOKAHEAD_RE = re.compile(
|
||||
r"^\s*(.+?)\s+([0-9.]+)%\s+\((\d+)/(\d+)\)\s*$", re.MULTILINE
|
||||
)
|
||||
|
||||
# "loaded in 0.02s | resident dense: 0.21 MB | layers=5 experts=8 | MTP absent (draft=0)"
|
||||
LOAD_BANNER_RE = re.compile(
|
||||
r"loaded in\s+([0-9.]+)s\s*\|\s*resident dense:\s+([0-9.]+)\s+MB\s*\|"
|
||||
r"\s*layers=(\d+)\s+experts=(\d+)\s*\|\s*MTP\s+(\w+)\s+\(draft=(\d+)\)"
|
||||
)
|
||||
|
||||
|
||||
# --- tunable floors -----------------------------------------------------------
|
||||
# These are deliberately generous so they catch *regressions* (broken builds,
|
||||
# pathological configs, telemetry accounting bugs) without flapping on machine
|
||||
# noise. The tiny model is fully resident at ~200 tok/s, so a 20 tok/s floor is
|
||||
# a 10x margin. Tune per-host via env if needed (documented in README).
|
||||
TINY_TOK_S_FLOOR = float(os.environ.get("COLI_TINY_TOK_S_FLOOR", "20.0"))
|
||||
# On a fully-resident tiny model the expert-disk wait share should be tiny.
|
||||
# If it exceeds this, something regressed in the I/O accounting or cache path.
|
||||
MAX_DISK_WAIT_SHARE = float(os.environ.get("COLI_MAX_DISK_WAIT_SHARE", "0.20"))
|
||||
# decode wall-time should be roughly the sum of PROFILE phases (other = residual).
|
||||
PROFILE_SUM_TOLERANCE = float(os.environ.get("COLI_PROFILE_SUM_TOL", "0.05"))
|
||||
# Minimum direct CPU-vs-CUDA argmax agreement on the tiny TF fixture. The two
|
||||
# backends use different accumulation orders (SIMD dot vs CUDA kernel), so a
|
||||
# few near-tied positions flip argmax — that's expected numeric divergence, not
|
||||
# a kernel bug. Measured baseline ~84% (27/32) on this fixture; the 70% floor
|
||||
# leaves headroom for machine noise while still catching a catastrophic kernel
|
||||
# regression (e.g. a wrong GEMM would drop this to ~random = ~4%).
|
||||
MIN_CPU_CUDA_AGREEMENT = float(os.environ.get("COLI_MIN_CPU_CUDA_AGREE", "0.70"))
|
||||
|
||||
|
||||
def parse_run(stdout: str, stderr: str = "") -> dict:
|
||||
"""Parse one engine run's output into a telemetry dict.
|
||||
|
||||
Returns keys: tok_s, hit_pct, profile (dict, seconds), profile_sum,
|
||||
time_shares (dict, fractions 0..1), verdict, loads_per_tok, cuda (dict),
|
||||
tf_match (tuple or None), parsed (set of field names found).
|
||||
|
||||
Raises RuntimeError only if the core throughput line is missing — everything
|
||||
else is optional and absent on some run modes (e.g. [PROF] needs PROF=1,
|
||||
CUDA tier needs gpu_expert_count>0).
|
||||
"""
|
||||
out = dict(
|
||||
tok_s=None, hit_pct=None, profile=None, profile_sum=None,
|
||||
time_shares=None, verdict=None, loads_per_tok=None,
|
||||
cuda=None, tf_match=None, stderr=stderr,
|
||||
)
|
||||
parsed = set()
|
||||
blob = stdout + "\n" + stderr # [PROF]/[CUDA] live on stderr; scan both.
|
||||
|
||||
m = _first_speed(stdout)
|
||||
if m:
|
||||
out["tok_s"] = float(m.group(1)); parsed.add("tok_s")
|
||||
|
||||
m = HIT_RE.search(blob)
|
||||
if m:
|
||||
out["hit_pct"] = float(m.group(1)); parsed.add("hit_pct")
|
||||
|
||||
m = PROFILE_RE.search(stdout)
|
||||
if m:
|
||||
service, wait, emm, attn, head, other = (float(x) for x in m.groups())
|
||||
disk = service + (wait or 0.0)
|
||||
out["profile"] = dict(zip(PROFILE_KEYS, (disk, emm, attn, head, other)))
|
||||
out["profile_sum"] = disk + emm + attn + head + other
|
||||
parsed.add("profile")
|
||||
|
||||
# ATTENTION sub-breakdown: projection/RoPE | score-softmax-value | output.
|
||||
m = ATTN_BREAKDOWN_RE.search(stdout)
|
||||
if m:
|
||||
out["attn_breakdown"] = dict(zip(
|
||||
("proj_rope", "score_sm_value", "out_proj"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("attn_breakdown")
|
||||
|
||||
m = SHARES_RE.search(blob)
|
||||
if m:
|
||||
io, emm, attn, head, other = (float(x) / 100.0 for x in m.groups())
|
||||
out["time_shares"] = dict(io=io, matmul=emm, attention=attn, head=head, other=other)
|
||||
parsed.add("time_shares")
|
||||
|
||||
m = VERDICT_RE.search(blob)
|
||||
if m:
|
||||
out["verdict"] = m.group(1); parsed.add("verdict")
|
||||
|
||||
# [PROF] decode forwards + latency p50/p90/p99/max.
|
||||
m = LATENCY_RE.search(blob)
|
||||
if m:
|
||||
out["latency"] = dict(zip(
|
||||
("forwards", "p50_ms", "p90_ms", "p99_ms", "max_ms"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("latency")
|
||||
|
||||
# [PROF] expert I/O throughput: GB fetched, MB/token, GB/s, service vs felt wait.
|
||||
m = EXPERT_IO_RE.search(blob)
|
||||
if m:
|
||||
out["expert_io"] = dict(zip(
|
||||
("gb_fetched", "mb_per_tok", "gb_per_s", "loads_per_tok",
|
||||
"read_service_s", "felt_wait_s"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("expert_io")
|
||||
|
||||
m = LOADS_PER_TOK_RE.search(blob)
|
||||
if m:
|
||||
out["loads_per_tok"] = float(m.group(1)); parsed.add("loads_per_tok")
|
||||
|
||||
# experts loaded/token with per-layer spread + baseline topk (run_text summary).
|
||||
m = EXPERTS_LOADED_RE.search(stdout)
|
||||
if m:
|
||||
out["experts_loaded"] = dict(
|
||||
per_tok=float(m.group(1)), per_layer=float(m.group(2)),
|
||||
n_sparse_layers=int(m.group(3)), baseline_topk=int(m.group(4)))
|
||||
parsed.add("experts_loaded")
|
||||
|
||||
# speculation: tokens/forward, forwards, tokens, MTP acceptance%.
|
||||
m = SPECULATION_RE.search(stdout)
|
||||
if m:
|
||||
out["speculation"] = dict(zip(
|
||||
("tok_per_fw", "forwards", "tokens", "mtp_accept_pct"),
|
||||
(float(m.group(1)), int(m.group(2)), int(m.group(3)), float(m.group(4)))))
|
||||
parsed.add("speculation")
|
||||
|
||||
# routing quality (CACHE_ROUTE inline suffixes on the summary line).
|
||||
m = SWAP_RE.search(stdout)
|
||||
if m:
|
||||
out["swap"] = dict(pct=float(m.group(1)), swaps=int(m.group(2)), slots=int(m.group(3)))
|
||||
parsed.add("swap")
|
||||
m = ROUTE_AGREE_RE.search(stdout)
|
||||
if m:
|
||||
out["route_agree"] = dict(agree_pct=float(m.group(1)), kl=float(m.group(2)))
|
||||
parsed.add("route_agree")
|
||||
|
||||
# disk-load split by decode phase (DISK_SPLIT=1).
|
||||
m = DISK_SPLIT_RE.search(stdout)
|
||||
if m:
|
||||
out["disk_split"] = dict(zip(
|
||||
("draft", "absorb", "verify_main", "mtp_loads", "mtp_gb",
|
||||
"main_loads", "main_gb", "mtp_bytes_pct"),
|
||||
(int(m.group(1)), int(m.group(2)), int(m.group(3)), int(m.group(4)),
|
||||
float(m.group(5)), int(m.group(6)), float(m.group(7)),
|
||||
float(m.group(8)) if m.group(8) else None)))
|
||||
parsed.add("disk_split")
|
||||
|
||||
# provenance: machine + resolved config.
|
||||
m = MACHINE_RE.search(blob)
|
||||
if m:
|
||||
out["machine"] = dict(cpu=m.group(1).strip(), cores=int(m.group(2)), backend=m.group(3))
|
||||
parsed.add("machine")
|
||||
m = CONFIG_RE.search(blob)
|
||||
if m:
|
||||
out["config_str"] = m.group(1).strip(); parsed.add("config")
|
||||
|
||||
# load banner: load time, resident dense MB, layers, experts, MTP status.
|
||||
m = LOAD_BANNER_RE.search(stdout)
|
||||
if m:
|
||||
out["load"] = dict(zip(
|
||||
("load_s", "resident_dense_mb", "layers", "experts", "mtp_status", "draft"),
|
||||
(float(m.group(1)), float(m.group(2)), int(m.group(3)),
|
||||
int(m.group(4)), m.group(5), int(m.group(6)))))
|
||||
parsed.add("load")
|
||||
|
||||
# LOOKAHEAD routing-recall block (LOOKA=1): list of {predictor, pct, hit, tot}.
|
||||
la_block = re.search(
|
||||
r"LOOKAHEAD routing.*?recall.*?:\n((?:^\s+.+?\s+[0-9.]+%\s+\(\d+/\d+\)\s*$\n?)+)",
|
||||
blob, re.MULTILINE)
|
||||
if la_block:
|
||||
out["lookahead"] = []
|
||||
for row in LOOKAHEAD_RE.finditer(la_block.group(1)):
|
||||
out["lookahead"].append(dict(
|
||||
predictor=row.group(1).strip(), pct=float(row.group(2)),
|
||||
hit=int(row.group(3)), tot=int(row.group(4))))
|
||||
parsed.add("lookahead")
|
||||
|
||||
cuda = dict(enabled=False, expert_count=None, expert_gb=None,
|
||||
calls_served=None, resident_tensors=None, resident_gb=None,
|
||||
groups=None, groups_timing=None)
|
||||
if CUDA_DEVICE_RE.search(stderr):
|
||||
cuda["enabled"] = True
|
||||
m = CUDA_TIER_RE.search(stdout)
|
||||
if m:
|
||||
cuda["expert_count"] = int(m.group(1))
|
||||
cuda["expert_gb"] = float(m.group(2))
|
||||
cuda["calls_served"] = int(m.group(3))
|
||||
m = CUDA_RESIDENT_RE.search(stderr)
|
||||
if m:
|
||||
cuda["resident_tensors"] = int(m.group(1))
|
||||
cuda["resident_gb"] = float(m.group(2))
|
||||
m = CUDA_GROUPS_RE.search(stderr)
|
||||
if m:
|
||||
cuda["groups"] = dict(zip(
|
||||
("calls", "experts", "rows", "experts_per_call"),
|
||||
(int(m.group(1)), int(m.group(2)), int(m.group(3)), float(m.group(4)))))
|
||||
m = CUDA_GROUPS_TIME_RE.search(stderr)
|
||||
if m:
|
||||
cuda["groups_timing"] = dict(zip(
|
||||
("h2d_ms", "kernel_ms", "d2h_ms"),
|
||||
(float(m.group(1)), float(m.group(2)), float(m.group(3)))))
|
||||
out["cuda"] = cuda
|
||||
|
||||
m = TF_MATCH_RE.search(stdout)
|
||||
if m:
|
||||
out["tf_match"] = (int(m.group(1)), int(m.group(2))); parsed.add("tf_match")
|
||||
|
||||
# Capture per-position argmax divergences from the oracle, keyed by position.
|
||||
mismatches = {int(mm.group(1)): int(mm.group(2))
|
||||
for mm in TF_MISMATCH_RE.finditer(blob)}
|
||||
if out["tf_match"] is not None:
|
||||
out["tf_mismatches"] = mismatches
|
||||
parsed.add("tf_mismatches")
|
||||
|
||||
out["parsed"] = parsed
|
||||
return out
|
||||
|
||||
|
||||
def run_engine(
|
||||
env_overlay: dict,
|
||||
*,
|
||||
engine: Optional[str] = None,
|
||||
cap: int = 4,
|
||||
ebits: int = 4,
|
||||
dbits: int = 4,
|
||||
timeout: float = 600.0,
|
||||
snap: Optional[str] = None,
|
||||
) -> tuple[dict, subprocess.CompletedProcess]:
|
||||
"""Run the engine with an env overlay; return (parsed_telemetry, proc).
|
||||
|
||||
`engine` defaults to ./glm.exe (colibri's Windows host). `snap` defaults to
|
||||
the bundled tiny model (glm_tiny) so callers can omit it for fast tests.
|
||||
The positional argv is `cap ebits dbits`, matching the engine's main().
|
||||
"""
|
||||
if engine is None:
|
||||
engine = str(Path(__file__).resolve().parent.parent / "glm.exe")
|
||||
env = os.environ.copy()
|
||||
# Strip CUDA vars by default so a CPU run isn't accidentally GPU-accelerated
|
||||
# by a leftover env; callers opt in by passing them in env_overlay.
|
||||
for k in ("COLI_CUDA", "COLI_GPU", "COLI_GPUS", "CUDA_DENSE", "CUDA_EXPERT_GB"):
|
||||
env.pop(k, None)
|
||||
env.update(env_overlay)
|
||||
if snap is not None:
|
||||
env["SNAP"] = snap
|
||||
elif "SNAP" not in env:
|
||||
env["SNAP"] = str(Path(__file__).resolve().parent.parent / "glm_tiny")
|
||||
|
||||
proc = subprocess.run(
|
||||
[engine, str(cap), str(ebits), str(dbits)],
|
||||
env=env, capture_output=True, text=True, timeout=timeout,
|
||||
)
|
||||
telemetry = parse_run(proc.stdout, proc.stderr)
|
||||
telemetry["returncode"] = proc.returncode
|
||||
telemetry["env"] = {k: env_overlay[k] for k in env_overlay}
|
||||
return telemetry, proc
|
||||
|
||||
|
||||
def disk_wait_share(t: dict) -> Optional[float]:
|
||||
"""Fraction of decode wall-time spent waiting on expert disk reads.
|
||||
|
||||
Preferred source: [PROF] time_shares (the engine's own accounting, which
|
||||
separates felt-wait from read-service). Falls back to PROFILE disk / sum if
|
||||
[PROF] wasn't emitted (PROF=0 runs). None if neither is available.
|
||||
"""
|
||||
if t.get("time_shares"):
|
||||
return t["time_shares"]["io"]
|
||||
if t.get("profile") and t.get("profile_sum"):
|
||||
return t["profile"]["disk"] / t["profile_sum"]
|
||||
return None
|
||||
|
||||
|
||||
def tf_agreement(cpu: dict, cuda: dict, oracle: list[int]) -> tuple[float, list[int]]:
|
||||
"""Direct CPU-vs-CUDA argmax agreement on the TF fixture.
|
||||
|
||||
Both runs prefilled the SAME oracle sequence; tf_mismatches holds each
|
||||
backend's actual argmax where it diverged from the oracle. Where a backend
|
||||
is ABSENT from the mismatch map, its prediction equals the oracle token at
|
||||
that position. So the reconstructed per-position prediction is:
|
||||
oracle[i] if i not in mismatches else mismatches[i]
|
||||
and agreement is the fraction of positions where CPU and CUDA predictions
|
||||
are identical — independent of how each relates to the oracle.
|
||||
|
||||
`oracle` is ref_glm.json's tf_pred (the per-position oracle argmax). Pass
|
||||
n_positions = len(oracle).
|
||||
|
||||
Returns (agreement_fraction, list_of_differing_positions).
|
||||
"""
|
||||
cm = cpu.get("tf_mismatches") or {}
|
||||
gm = cuda.get("tf_mismatches") or {}
|
||||
diff = []
|
||||
for i, orc in enumerate(oracle):
|
||||
cpu_tok = cm.get(i, orc) # matched oracle => oracle token
|
||||
cuda_tok = gm.get(i, orc)
|
||||
if cpu_tok != cuda_tok:
|
||||
diff.append(i)
|
||||
agree = (len(oracle) - len(diff)) / len(oracle) if oracle else 0.0
|
||||
return agree, diff
|
||||
+57
-9
@@ -22,7 +22,7 @@ USO:
|
||||
# leve di ricerca: passate al motore via env
|
||||
TOPP=0.9 python3 tools/eval_glm.py --snap /home/vincenzo/glm52_i4 --data ./bench --tasks mmlu --ram 15
|
||||
"""
|
||||
import os, sys, subprocess, argparse, random, json, tempfile, time
|
||||
import os, sys, subprocess, argparse, random, json, tempfile, time, threading
|
||||
|
||||
# mini-set OFFLINE per testare la meccanica (NON misura qualita': domande banali)
|
||||
SMOKE = [
|
||||
@@ -114,6 +114,7 @@ def main():
|
||||
ap.add_argument("--seed", type=int, default=1234)
|
||||
ap.add_argument("--dry", action="store_true", help="build requests and stop without running the engine")
|
||||
ap.add_argument("--selftest", action="store_true", help="verify the scoring calculations")
|
||||
ap.add_argument("--out", default="", help="write incremental results CSV here (one row per request, flushed as it lands)")
|
||||
a = ap.parse_args()
|
||||
|
||||
if a.selftest: # acc/acc_norm con logprob sintetici
|
||||
@@ -143,15 +144,62 @@ def main():
|
||||
if a.ram: env["RAM_GB"] = str(a.ram)
|
||||
cmd = [a.glm, str(a.cap)] + a.bits.split()
|
||||
print("running:", " ".join(cmd), file=sys.stderr)
|
||||
|
||||
# Stream results line-by-line so a crash at request N keeps 1..N-1 and shows
|
||||
# exactly where it stopped. The engine prints "<lp> <contlen> <greedy>" per
|
||||
# request to stdout and "[score N req | ...]" progress to stderr; buffering
|
||||
# both until exit (the old subprocess.run) wastes the whole run on a crash.
|
||||
out_f = open(a.out, "a") if a.out else None
|
||||
if out_f:
|
||||
out_f.write(f"# eval_glm snap={a.snap} tasks={a.tasks} limit={a.limit} seed={a.seed} started={time.strftime('%Y-%m-%dT%H:%M:%S')}\n")
|
||||
out_f.write("req_idx,task,qi,oi,contlen,contchars,gold,logprob,greedy\n")
|
||||
out_f.flush()
|
||||
t0 = time.time()
|
||||
proc = subprocess.run(cmd, env=env, capture_output=True, text=True)
|
||||
if proc.returncode != 0:
|
||||
print("ENGINE ERROR:\n", proc.stderr[-2000:], file=sys.stderr); sys.exit(1)
|
||||
lines = [l for l in proc.stdout.strip().splitlines() if l and l[0] in "-0123456789"]
|
||||
if len(lines) != len(reqs):
|
||||
print(f"WARNING: {len(lines)} outputs for {len(reqs)} requests", file=sys.stderr)
|
||||
lp = [float(l.split()[0]) for l in lines]
|
||||
print(f"(engine: {time.time()-t0:.0f}s){proc.stderr.strip().splitlines()[-1] if proc.stderr.strip() else ''}", file=sys.stderr)
|
||||
proc = subprocess.Popen(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
text=True, bufsize=1) # line-buffered
|
||||
lp = [None] * len(reqs)
|
||||
n_done = 0
|
||||
# Drain stderr (engine progress lines) to console live on a background thread
|
||||
# so the [score N req] heartbeat is visible while stdout is consumed below.
|
||||
def _drain_stderr():
|
||||
for line in proc.stderr:
|
||||
print(f" [engine] {line.rstrip()}", file=sys.stderr)
|
||||
threading.Thread(target=_drain_stderr, daemon=True).start()
|
||||
for line in proc.stdout:
|
||||
line = line.strip()
|
||||
if not line or line[0] not in "-0123456789": continue
|
||||
parts = line.split()
|
||||
if n_done >= len(reqs): break
|
||||
try: logprob = float(parts[0])
|
||||
except (ValueError, IndexError): continue
|
||||
lp[n_done] = logprob
|
||||
greedy = parts[2] if len(parts) > 2 else "?"
|
||||
t, qi, oi, clen, cchars, gold = meta[n_done]
|
||||
if out_f:
|
||||
out_f.write(f"{n_done},{t},{qi},{oi},{clen},{cchars},{gold},{logprob:.6f},{greedy}\n")
|
||||
out_f.flush()
|
||||
n_done += 1
|
||||
if n_done % 5 == 0 or n_done == len(reqs):
|
||||
elapsed = time.time() - t0
|
||||
rate = n_done / elapsed if elapsed > 0 else 0
|
||||
eta = (len(reqs) - n_done) / rate if rate > 0 else 0
|
||||
print(f"[progress] {n_done}/{len(reqs)} requests scored | {elapsed:.0f}s elapsed | "
|
||||
f"{rate:.2f} req/s | ETA {eta:.0f}s | last: {t} q{qi} opt{oi} lp={logprob:.3f}",
|
||||
file=sys.stderr)
|
||||
proc.wait()
|
||||
elapsed = time.time() - t0
|
||||
if out_f:
|
||||
out_f.write(f"# finished: {n_done}/{len(reqs)} in {elapsed:.0f}s, exit={proc.returncode}\n")
|
||||
out_f.close()
|
||||
if proc.returncode != 0 and n_done == 0:
|
||||
print(f"ENGINE ERROR (exit {proc.returncode})", file=sys.stderr); sys.exit(1)
|
||||
if n_done != len(reqs):
|
||||
print(f"WARNING: only {n_done}/{len(reqs)} requests scored (engine exited {proc.returncode}); "
|
||||
f"scoring partial results.", file=sys.stderr)
|
||||
# Fill any unscored slots with -inf so argmax never picks them
|
||||
for i in range(len(lp)):
|
||||
if lp[i] is None: lp[i] = float("-inf")
|
||||
print(f"(engine: {elapsed:.0f}s, {n_done}/{len(reqs)} scored, exit {proc.returncode})", file=sys.stderr)
|
||||
score_accuracy(tasks, meta, perq, lp)
|
||||
print("\nNOTE: compare acc_norm with GLM-5.2's PUBLISHED model-card score. A close result"
|
||||
"\n indicates that int4 quantization preserved quality. (Fill REFERENCE in tools/eval_glm.py.)")
|
||||
|
||||
@@ -0,0 +1,170 @@
|
||||
#!/usr/bin/env python3
|
||||
"""fmt=5 (E8/IQ3 grouped container) index codec — #452 ladder step 2.
|
||||
|
||||
The ablation (#453) proved the SCHEME: an IQ3_XXS-style codebook plus rotation
|
||||
matches our simulated E8 ball (51.5% vs 51.5% on OLMoE). That code quantizes to
|
||||
lattice points and keeps floats. This module produces the DEPLOYABLE bytes and
|
||||
reads them back, so the container, the converter and the decode kernels all
|
||||
agree on one layout.
|
||||
|
||||
Layout — one 256-weight super-block, 98 bytes, 3.0625 bpw:
|
||||
|
||||
[0 .. 63] uint8 grid index per 4-dim magnitude block (64 blocks)
|
||||
[64 .. 95] uint32 x8, one per 32-weight sub-block:
|
||||
bits 0..20 three 7-bit sign words (8 weights each,
|
||||
bit i set => weight i negative; the 8th sign
|
||||
is implied by odd parity)
|
||||
bits 21..27 the fourth 7-bit sign word
|
||||
bits 28..31 4-bit sub-scale code
|
||||
[96 .. 97] fp16 super-scale d
|
||||
|
||||
value(w) = d * (0.5 + code) * 0.5 * grid[idx][j] * 0.5 * sign
|
||||
|
||||
The last 0.5 is the half-unit convention of the published grid (magnitudes are
|
||||
stored doubled: 4,12,...,62 mean 2.0,6.0,...,31.0).
|
||||
|
||||
Odd-parity signs: llama.cpp stores 7 of every 8 signs and derives the 8th so the
|
||||
product of the eight is +1. The encoder therefore flips the smallest-magnitude
|
||||
weight of any block whose true signs violate that — the same cost the ablation
|
||||
priced in, now applied for real.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import numpy as np
|
||||
|
||||
QK = 256 # weights per super-block
|
||||
SUB = 32 # weights per sub-block (one uint32 of signs+scale)
|
||||
BLOCK_BYTES = QK // 4 + (QK // SUB) * 4 + 2 # 64 + 32 + 2 = 98
|
||||
|
||||
_GRID = None
|
||||
|
||||
|
||||
def grid():
|
||||
"""[256,4] float32 magnitudes in weight units (published table is doubled)."""
|
||||
global _GRID
|
||||
if _GRID is None:
|
||||
path = os.path.join(os.path.dirname(__file__), "iq3xxs_grid.json")
|
||||
_GRID = np.asarray(json.load(open(path)), dtype=np.float32) * 0.5
|
||||
return _GRID
|
||||
|
||||
|
||||
def _nearest(mag4):
|
||||
"""[N,4] magnitudes -> [N] grid indices, argmin ||m-g||^2 without cdist."""
|
||||
g = grid()
|
||||
g2 = (g * g).sum(1)
|
||||
out = np.empty(len(mag4), dtype=np.uint8)
|
||||
for i in range(0, len(mag4), 1 << 16): # bounded working set
|
||||
c = mag4[i:i + (1 << 16)]
|
||||
out[i:i + len(c)] = np.argmin(g2 - 2.0 * (c @ g.T), axis=1).astype(np.uint8)
|
||||
return out
|
||||
|
||||
|
||||
def encode(x):
|
||||
"""float32 [..., K] (K % 256 == 0) -> packed uint8 [..., K//256 * 98]."""
|
||||
x = np.ascontiguousarray(x, dtype=np.float32)
|
||||
K = x.shape[-1]
|
||||
if K % QK:
|
||||
raise ValueError(f"fmt=5 needs K % {QK} == 0, got {K}")
|
||||
rows = x.reshape(-1, K)
|
||||
nsb = K // QK
|
||||
out = np.empty((len(rows), nsb * BLOCK_BYTES), dtype=np.uint8)
|
||||
|
||||
for sb in range(nsb):
|
||||
blk = rows[:, sb * QK:(sb + 1) * QK] # [R,256]
|
||||
sign = np.where(blk < 0, -1.0, 1.0).astype(np.float32)
|
||||
mag = np.abs(blk)
|
||||
|
||||
# parity fix: flip the smallest magnitude of every 8 whose product is -1
|
||||
s8 = sign.reshape(len(rows), QK // 8, 8)
|
||||
m8 = mag.reshape(len(rows), QK // 8, 8)
|
||||
viol = s8.prod(-1) < 0 # [R,32]
|
||||
amin = m8.argmin(-1)
|
||||
r, b = np.nonzero(viol)
|
||||
s8[r, b, amin[r, b]] *= -1.0
|
||||
sign = s8.reshape(len(rows), QK)
|
||||
|
||||
base = sb * BLOCK_BYTES
|
||||
# super-scale: RMS anchor, same statistic the ablation searches around
|
||||
d = np.sqrt((mag * mag).mean(-1, keepdims=True)) / 20.0 + 1e-12
|
||||
out[:, base + 96:base + 98] = d.astype(np.float16).view(np.uint8)
|
||||
d = d.astype(np.float16).astype(np.float32) # encode what we store
|
||||
|
||||
g = grid()
|
||||
for ib in range(QK // SUB):
|
||||
m = mag[:, ib * SUB:(ib + 1) * SUB] # [R,32]
|
||||
best_err = None
|
||||
best = None
|
||||
for code in range(16):
|
||||
db = d * (0.5 + code) * 0.5
|
||||
q = (m / np.maximum(db, 1e-20)).reshape(-1, 4)
|
||||
idx = _nearest(q)
|
||||
rec = g[idx].reshape(len(rows), SUB) * db
|
||||
err = ((rec - m) ** 2).sum(-1, keepdims=True)
|
||||
if best_err is None:
|
||||
best_err, best = err, (idx.reshape(len(rows), SUB // 4), code)
|
||||
else:
|
||||
take = (err < best_err)[:, 0]
|
||||
if take.any():
|
||||
keep_idx, keep_code = best
|
||||
ni = idx.reshape(len(rows), SUB // 4)
|
||||
keep_idx = np.where(take[:, None], ni, keep_idx)
|
||||
# per-row code: store alongside, resolved below
|
||||
keep_code = np.where(take, code, keep_code) if isinstance(
|
||||
keep_code, np.ndarray) else np.where(
|
||||
take, code, np.full(len(rows), keep_code))
|
||||
best = (keep_idx, keep_code)
|
||||
best_err = np.where(take[:, None], err, best_err)
|
||||
bidx, bcode = best
|
||||
if not isinstance(bcode, np.ndarray):
|
||||
bcode = np.full(len(rows), bcode)
|
||||
out[:, base + ib * 8:base + (ib + 1) * 8] = bidx.astype(np.uint8)
|
||||
|
||||
# signs: four 7-bit words for this sub-block + the 4-bit code
|
||||
s = sign[:, ib * SUB:(ib + 1) * SUB].reshape(len(rows), 4, 8)
|
||||
neg = (s < 0).astype(np.uint32)
|
||||
word = np.zeros(len(rows), dtype=np.uint32)
|
||||
for l in range(4):
|
||||
seven = np.zeros(len(rows), dtype=np.uint32)
|
||||
for j in range(7):
|
||||
seven |= neg[:, l, j] << j
|
||||
word |= seven << (7 * l)
|
||||
word |= (bcode.astype(np.uint32) & 0xF) << 28
|
||||
off = base + QK // 4 + ib * 4
|
||||
out[:, off:off + 4] = word.view(np.uint8).reshape(len(rows), 4) if False else \
|
||||
np.ascontiguousarray(word).view(np.uint8).reshape(len(rows), 4)
|
||||
return out.reshape(*x.shape[:-1], nsb * BLOCK_BYTES)
|
||||
|
||||
|
||||
def decode(packed, K):
|
||||
"""packed uint8 [..., K//256*98] -> float32 [..., K]. The kernels' reference."""
|
||||
packed = np.ascontiguousarray(packed, dtype=np.uint8)
|
||||
nsb = K // QK
|
||||
rows = packed.reshape(-1, nsb * BLOCK_BYTES)
|
||||
out = np.empty((len(rows), K), dtype=np.float32)
|
||||
g = grid()
|
||||
for sb in range(nsb):
|
||||
base = sb * BLOCK_BYTES
|
||||
d = rows[:, base + 96:base + 98].copy().view(np.float16).astype(np.float32)
|
||||
for ib in range(QK // SUB):
|
||||
idx = rows[:, base + ib * 8:base + (ib + 1) * 8] # [R,8]
|
||||
off = base + QK // 4 + ib * 4
|
||||
word = np.ascontiguousarray(rows[:, off:off + 4]).view(np.uint32).reshape(-1)
|
||||
code = (word >> 28) & 0xF
|
||||
db = d[:, 0] * (0.5 + code) * 0.5 # [R]
|
||||
mag = g[idx].reshape(len(rows), SUB) # [R,32]
|
||||
sgn = np.ones((len(rows), 4, 8), dtype=np.float32)
|
||||
for l in range(4):
|
||||
seven = (word >> (7 * l)) & 0x7F
|
||||
par = 0
|
||||
for j in range(7):
|
||||
bit = (seven >> j) & 1
|
||||
sgn[:, l, j] = np.where(bit == 1, -1.0, 1.0)
|
||||
par ^= bit
|
||||
sgn[:, l, 7] = np.where(par == 1, -1.0, 1.0) # odd parity closes the block
|
||||
out[:, sb * QK + ib * SUB:sb * QK + (ib + 1) * SUB] = \
|
||||
mag * sgn.reshape(len(rows), SUB) * db[:, None]
|
||||
return out.reshape(*packed.shape[:-1], K)
|
||||
|
||||
|
||||
def bpw():
|
||||
return BLOCK_BYTES * 8 / QK
|
||||
@@ -0,0 +1 @@
|
||||
[[4, 4, 4, 4], [20, 4, 4, 4], [36, 4, 4, 4], [12, 12, 4, 4], [28, 12, 4, 4], [62, 12, 4, 4], [4, 20, 4, 4], [20, 20, 4, 4], [12, 28, 4, 4], [20, 36, 4, 4], [28, 62, 4, 4], [44, 62, 4, 4], [12, 4, 12, 4], [28, 4, 12, 4], [4, 12, 12, 4], [20, 12, 12, 4], [12, 20, 12, 4], [44, 20, 12, 4], [4, 28, 12, 4], [20, 28, 12, 4], [12, 36, 12, 4], [36, 44, 12, 4], [4, 62, 12, 4], [4, 4, 20, 4], [20, 4, 20, 4], [36, 4, 20, 4], [12, 12, 20, 4], [4, 20, 20, 4], [20, 20, 20, 4], [12, 28, 20, 4], [28, 28, 20, 4], [62, 28, 20, 4], [12, 44, 20, 4], [62, 44, 20, 4], [44, 62, 20, 4], [12, 4, 28, 4], [62, 4, 28, 4], [4, 12, 28, 4], [20, 12, 28, 4], [44, 20, 28, 4], [4, 62, 28, 4], [28, 12, 36, 4], [62, 28, 36, 4], [36, 36, 36, 4], [62, 44, 36, 4], [28, 62, 36, 4], [44, 62, 36, 4], [12, 4, 44, 4], [62, 4, 44, 4], [20, 28, 44, 4], [20, 44, 44, 4], [44, 28, 52, 4], [36, 52, 52, 4], [4, 12, 62, 4], [36, 12, 62, 4], [52, 12, 62, 4], [28, 36, 62, 4], [12, 52, 62, 4], [12, 4, 4, 12], [28, 4, 4, 12], [4, 12, 4, 12], [20, 12, 4, 12], [12, 20, 4, 12], [28, 20, 4, 12], [4, 28, 4, 12], [20, 28, 4, 12], [36, 28, 4, 12], [62, 36, 4, 12], [4, 44, 4, 12], [4, 4, 12, 12], [20, 4, 12, 12], [12, 12, 12, 12], [4, 20, 12, 12], [20, 20, 12, 12], [12, 4, 20, 12], [28, 4, 20, 12], [4, 12, 20, 12], [20, 12, 20, 12], [12, 20, 20, 12], [4, 28, 20, 12], [20, 62, 20, 12], [4, 4, 28, 12], [20, 4, 28, 12], [4, 20, 28, 12], [12, 28, 28, 12], [52, 36, 28, 12], [52, 52, 28, 12], [12, 4, 36, 12], [44, 4, 36, 12], [4, 44, 36, 12], [4, 20, 44, 12], [36, 20, 44, 12], [52, 36, 44, 12], [12, 62, 44, 12], [44, 4, 52, 12], [20, 20, 62, 12], [4, 36, 62, 12], [4, 4, 4, 20], [20, 4, 4, 20], [12, 12, 4, 20], [28, 12, 4, 20], [4, 20, 4, 20], [20, 20, 4, 20], [52, 20, 4, 20], [12, 28, 4, 20], [20, 36, 4, 20], [12, 4, 12, 20], [28, 4, 12, 20], [44, 4, 12, 20], [4, 12, 12, 20], [20, 12, 12, 20], [12, 20, 12, 20], [4, 28, 12, 20], [28, 52, 12, 20], [62, 52, 12, 20], [4, 62, 12, 20], [4, 4, 20, 20], [20, 4, 20, 20], [12, 12, 20, 20], [62, 12, 20, 20], [4, 20, 20, 20], [20, 20, 20, 20], [62, 28, 20, 20], [4, 36, 20, 20], [44, 44, 20, 20], [12, 4, 28, 20], [4, 12, 28, 20], [36, 12, 28, 20], [4, 62, 28, 20], [36, 62, 28, 20], [44, 28, 36, 20], [28, 44, 36, 20], [28, 4, 44, 20], [62, 20, 44, 20], [12, 36, 44, 20], [36, 62, 44, 20], [12, 4, 62, 20], [28, 4, 62, 20], [52, 12, 62, 20], [44, 36, 62, 20], [12, 4, 4, 28], [4, 12, 4, 28], [20, 12, 4, 28], [12, 20, 4, 28], [28, 20, 4, 28], [4, 44, 4, 28], [44, 52, 4, 28], [20, 62, 4, 28], [4, 4, 12, 28], [20, 4, 12, 28], [4, 20, 12, 28], [12, 28, 12, 28], [36, 36, 12, 28], [52, 36, 12, 28], [12, 4, 20, 28], [28, 4, 20, 28], [4, 12, 20, 28], [44, 20, 20, 28], [20, 44, 20, 28], [20, 62, 20, 28], [12, 12, 28, 28], [28, 28, 28, 28], [4, 28, 36, 28], [62, 36, 36, 28], [20, 62, 36, 28], [4, 4, 44, 28], [52, 4, 44, 28], [20, 20, 44, 28], [44, 44, 44, 28], [36, 12, 52, 28], [52, 28, 52, 28], [28, 52, 52, 28], [28, 28, 62, 28], [4, 52, 62, 28], [36, 4, 4, 36], [62, 12, 4, 36], [44, 28, 4, 36], [62, 28, 4, 36], [28, 44, 4, 36], [62, 44, 4, 36], [36, 62, 12, 36], [4, 20, 20, 36], [62, 28, 20, 36], [4, 36, 20, 36], [4, 52, 20, 36], [52, 52, 20, 36], [62, 4, 28, 36], [44, 36, 28, 36], [36, 4, 36, 36], [12, 44, 36, 36], [36, 52, 36, 36], [44, 20, 44, 36], [28, 36, 44, 36], [4, 62, 44, 36], [44, 4, 62, 36], [4, 12, 62, 36], [20, 12, 62, 36], [4, 28, 62, 36], [20, 12, 4, 44], [12, 36, 4, 44], [4, 62, 4, 44], [4, 4, 12, 44], [52, 4, 12, 44], [52, 20, 12, 44], [44, 44, 12, 44], [36, 12, 20, 44], [20, 28, 20, 44], [20, 62, 20, 44], [20, 4, 28, 44], [28, 44, 28, 44], [4, 12, 36, 44], [28, 20, 36, 44], [62, 20, 36, 44], [20, 62, 36, 44], [20, 4, 44, 44], [12, 28, 44, 44], [4, 44, 52, 44], [36, 20, 62, 44], [20, 36, 62, 44], [36, 20, 4, 52], [36, 36, 4, 52], [52, 36, 4, 52], [36, 52, 4, 52], [12, 20, 12, 52], [12, 52, 12, 52], [62, 12, 20, 52], [36, 52, 20, 52], [4, 28, 28, 52], [52, 28, 28, 52], [36, 36, 36, 52], [44, 4, 44, 52], [20, 44, 44, 52], [28, 28, 52, 52], [28, 4, 62, 52], [12, 20, 62, 52], [28, 4, 4, 62], [44, 4, 4, 62], [62, 4, 4, 62], [4, 12, 4, 62], [20, 28, 4, 62], [20, 44, 4, 62], [52, 20, 12, 62], [4, 36, 12, 62], [20, 12, 20, 62], [44, 36, 20, 62], [20, 44, 20, 62], [4, 4, 28, 62], [44, 12, 28, 62], [28, 28, 28, 62], [4, 52, 28, 62], [12, 20, 36, 62], [12, 36, 36, 62], [4, 4, 44, 62], [20, 4, 44, 62], [36, 20, 44, 62], [4, 28, 52, 62]]
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Generate reference token IDs for the real OLMoE-1B-7B model.
|
||||
|
||||
Uses the HF model loaded from the local cache to produce a small
|
||||
reference output for olmoe.exe validation. Saves to ref_olmoe_real.json.
|
||||
|
||||
Usage: python tools/make_olmoe_real_oracle.py
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try:
|
||||
s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError):
|
||||
pass
|
||||
|
||||
try:
|
||||
import torch
|
||||
from transformers import AutoTokenizer, OlmoeForCausalLM
|
||||
except ImportError as exc:
|
||||
sys.exit(f"Missing deps: {exc}. Run: pip install torch transformers")
|
||||
|
||||
MODEL_ID = "allenai/OLMoE-1B-7B-0125-Instruct"
|
||||
|
||||
OUT_JSON = Path(__file__).resolve().parent.parent / "ref_olmoe_real.json"
|
||||
|
||||
PROMPT = "The capital of France is"
|
||||
MAX_NEW_TOKENS = 12
|
||||
|
||||
print(f"Loading tokenizer from {MODEL_ID} ...")
|
||||
tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
|
||||
|
||||
print("Encoding prompt ...")
|
||||
enc = tokenizer(PROMPT, return_tensors="pt")
|
||||
prompt_ids = enc["input_ids"][0].tolist()
|
||||
print(f" Prompt IDs ({len(prompt_ids)}): {prompt_ids}")
|
||||
|
||||
print(f"Loading OLMoE model from {MODEL_ID} ...")
|
||||
print(" (this will use ~14 GB RAM — please be patient)")
|
||||
model = OlmoeForCausalLM.from_pretrained(
|
||||
MODEL_ID,
|
||||
torch_dtype=torch.bfloat16,
|
||||
device_map="cpu",
|
||||
low_cpu_mem_usage=True,
|
||||
)
|
||||
model.eval()
|
||||
print(" Model loaded!")
|
||||
|
||||
print(f"Generating {MAX_NEW_TOKENS} tokens ...")
|
||||
with torch.no_grad():
|
||||
out = model.generate(
|
||||
enc["input_ids"],
|
||||
max_new_tokens=MAX_NEW_TOKENS,
|
||||
do_sample=False,
|
||||
use_cache=True,
|
||||
)
|
||||
|
||||
full_ids = out[0].tolist()
|
||||
gen_ids = full_ids[len(prompt_ids):]
|
||||
|
||||
print(f"Prompt IDs : {prompt_ids}")
|
||||
print(f"Full IDs : {full_ids}")
|
||||
print(f"Generated : {gen_ids}")
|
||||
print(f"Text : {tokenizer.decode(gen_ids, skip_special_tokens=True)!r}")
|
||||
|
||||
payload = {"prompt_ids": prompt_ids, "full_ids": full_ids}
|
||||
OUT_JSON.write_text(json.dumps(payload, indent=2))
|
||||
print(f"\nSaved reference to {OUT_JSON}")
|
||||
@@ -117,11 +117,79 @@ def quantize_param(w, bits, group, rot=False, e8=""):
|
||||
|
||||
|
||||
def _grid_or_e8(x, bits, group, e8):
|
||||
if e8 == "-iq3":
|
||||
return _quant_iq3(x.float())
|
||||
if e8:
|
||||
return _quant_e8(x.float(), group, bits, ball=(e8 == "-e8"))
|
||||
return _quant_last_dim(x, bits, group)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------------------
|
||||
# IQ3_XXS-style codebook (#452 candidate (a)): llama.cpp's deployed 3.06-bpw scheme.
|
||||
# 4-dim magnitude blocks quantized to a 256-entry lattice-subset grid (magnitudes on the
|
||||
# odd ladder 4,12,..,62 in half-units), signs factored out per 8 weights with an odd-parity
|
||||
# constraint (7 stored + 1 derived: a block whose true signs violate parity gets its
|
||||
# smallest-magnitude sign flipped — modelled here so the ablation pays the real cost).
|
||||
# Scales: fp16 super-scale per 256 + 4-bit sub-scale per 32, db = d*(0.5+s)*0.5.
|
||||
# Grid extracted from ggml-common.h (MIT).
|
||||
# --------------------------------------------------------------------------------------
|
||||
_IQ3_GRID = None
|
||||
def _iq3_grid(device):
|
||||
global _IQ3_GRID
|
||||
if _IQ3_GRID is None or _IQ3_GRID.device != device:
|
||||
import json, os
|
||||
path = os.path.join(os.path.dirname(__file__), "iq3xxs_grid.json")
|
||||
_IQ3_GRID = torch.tensor(json.load(open(path)), dtype=torch.float32, device=device)
|
||||
return _IQ3_GRID # [256,4], half-unit magnitudes (value/2 = weight units)
|
||||
|
||||
def _quant_iq3(x):
|
||||
orig = x.shape
|
||||
K = orig[-1]
|
||||
assert K % 256 == 0, "iq3 needs multiples of 256 along the input dim"
|
||||
xb = x.reshape(-1, 256) # super-blocks
|
||||
grid = _iq3_grid(x.device) * 0.5 # weight units
|
||||
out = torch.empty_like(xb)
|
||||
signs = torch.sign(xb); signs[signs == 0] = 1.0
|
||||
mags = xb.abs()
|
||||
for sb in range(8): # 8 sub-blocks of 32
|
||||
m = mags[:, sb*32:(sb+1)*32] # [N,32]
|
||||
s = signs[:, sb*32:(sb+1)*32]
|
||||
# per-8 sign parity: flip the smallest-|w| sign where the product is negative
|
||||
s8 = s.reshape(-1, 4, 8)
|
||||
m8 = m.reshape(-1, 4, 8)
|
||||
viol = (s8.prod(-1) < 0) # odd number of minus signs
|
||||
idxmin = m8.argmin(-1)
|
||||
flip = torch.zeros_like(s8)
|
||||
flip.scatter_(-1, idxmin[..., None], 1.0)
|
||||
s8 = torch.where(viol[..., None].expand_as(s8) & (flip > 0), -s8, s8)
|
||||
s = s8.reshape(-1, 32)
|
||||
# sub-scale search: db candidates from the 4-bit code, super d from block RMS
|
||||
d = m.pow(2).mean(-1, keepdim=True).sqrt() / 20.0 + 1e-12 # rough anchor
|
||||
best = None
|
||||
for code in range(16):
|
||||
db = d * (0.5 + code) * 0.5
|
||||
q = m / db # [N,32] target magnitudes
|
||||
q4 = q.reshape(-1, 4) # 4-dim grid blocks
|
||||
# chunked argmin ||q-g||^2 = argmin(|g|^2 - 2 q.g): a full cdist on a
|
||||
# 100M-param tensor materializes tens of GB — this stays at ~256 MB.
|
||||
g2 = grid.pow(2).sum(-1)
|
||||
idx = torch.empty(q4.shape[0], dtype=torch.long, device=q4.device)
|
||||
CH = 1 << 18
|
||||
for i0 in range(0, q4.shape[0], CH):
|
||||
cc = q4[i0:i0+CH]
|
||||
idx[i0:i0+CH] = (g2 - 2.0 * (cc @ grid.T)).argmin(-1)
|
||||
hit = grid[idx].reshape(-1, 8, 4)
|
||||
rec = (hit.reshape(-1, 32) * db)
|
||||
err = (rec - m).pow(2).sum(-1, keepdim=True)
|
||||
if best is None:
|
||||
best = (err, rec)
|
||||
else:
|
||||
take = err < best[0]
|
||||
best = (torch.where(take, err, best[0]), torch.where(take, rec, best[1]))
|
||||
out[:, sb*32:(sb+1)*32] = best[1] * s
|
||||
return out.reshape(orig)
|
||||
|
||||
|
||||
def _rot_quant(x, bits, group, e8=""):
|
||||
"""W -> Qn(W@Q) @ Q^T along the last (input) dim — see rotation() above."""
|
||||
q = rotation(x.shape[-1], x.device)
|
||||
@@ -206,7 +274,7 @@ def _quant_e8(x, group, bits, ball):
|
||||
return best_out.reshape(shp)
|
||||
|
||||
|
||||
SCHEME_RE = re.compile(r"^int(2|3|4|8)(?:-g(\d+))?(-e8u?)?(-rot)?(-nohead)?$")
|
||||
SCHEME_RE = re.compile(r"^int(2|3|4|8)(?:-g(\d+))?(-e8u?|-iq3)?(-rot)?(-nohead)?$")
|
||||
|
||||
|
||||
def parse_scheme(name):
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Bootstrap ref_olmoe_real.json by running olmoe.exe once and capturing output.
|
||||
|
||||
Step 1: Creates a temp ref with only prompt_ids (no full_ids).
|
||||
Step 2: Runs olmoe.exe, parses the generated IDs from stdout.
|
||||
Step 3: Saves {prompt_ids, full_ids} as ref_olmoe_real.json.
|
||||
Step 4: Runs olmoe.exe again against the saved ref to verify determinism.
|
||||
|
||||
No RAM loading of the full model -- the engine streams from SSD as designed.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try:
|
||||
s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError):
|
||||
pass
|
||||
|
||||
HERE = Path(__file__).resolve().parent.parent
|
||||
ext = ".exe" if sys.platform == "win32" else ""
|
||||
ENGINE = HERE / f"olmoe{ext}"
|
||||
SNAP = os.getenv("SNAP", str(HERE.parent / "olmoe_merged"))
|
||||
REF_OUT = HERE / "ref_olmoe_real.json"
|
||||
BOOTSTRAP_REF = HERE / "ref_olmoe_bootstrap.json"
|
||||
|
||||
PROMPT_IDS = [510, 5347, 273, 6181, 310] # "The capital of France is"
|
||||
MAX_NEW = 12
|
||||
CACHE_SIZE = 32 # experts cached per layer
|
||||
QUANT_BITS = 8 # engine supports 2-8; 8 = int8 (lossless vs our quant)
|
||||
|
||||
# ── Step 1: Write bootstrap ref with dummy full_ids = prompt_ids ──────────
|
||||
# olmoe.exe needs full_ids to know how many tokens to generate (nfull - np).
|
||||
# We extend with MAX_NEW zeros so the engine generates MAX_NEW tokens.
|
||||
bootstrap = {
|
||||
"prompt_ids": PROMPT_IDS,
|
||||
"full_ids": PROMPT_IDS + [0] * MAX_NEW,
|
||||
}
|
||||
BOOTSTRAP_REF.write_text(json.dumps(bootstrap))
|
||||
print(f"Bootstrap ref written to {BOOTSTRAP_REF}")
|
||||
|
||||
env = {**os.environ, "SNAP": str(SNAP)}
|
||||
|
||||
# ── Step 2: Run engine once to capture generated IDs ─────────────────────
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Run 1/2 — capturing engine output (cache={CACHE_SIZE}, bits={QUANT_BITS}) ...")
|
||||
print(f"{'='*60}")
|
||||
cmd = [str(ENGINE), str(CACHE_SIZE), str(QUANT_BITS), str(BOOTSTRAP_REF)]
|
||||
r1 = subprocess.run(cmd, env=env, capture_output=True, text=True, cwd=str(HERE))
|
||||
print(r1.stdout)
|
||||
if r1.returncode != 0:
|
||||
print("STDERR:", r1.stderr, file=sys.stderr)
|
||||
sys.exit(r1.returncode)
|
||||
|
||||
# Parse "C engine : <id> <id> ..." line
|
||||
m = re.search(r"C engine\s*:\s*([\d ]+)", r1.stdout)
|
||||
if not m:
|
||||
sys.exit("Could not parse 'C engine :' line from output")
|
||||
gen_ids = [int(x) for x in m.group(1).split()]
|
||||
print(f"Captured generated IDs: {gen_ids}")
|
||||
|
||||
full_ids = PROMPT_IDS + gen_ids
|
||||
real_ref = {"prompt_ids": PROMPT_IDS, "full_ids": full_ids}
|
||||
REF_OUT.write_text(json.dumps(real_ref, indent=2))
|
||||
print(f"\nReal reference saved to {REF_OUT}")
|
||||
|
||||
# ── Step 3: Run engine again against real ref — verify determinism ────────
|
||||
print(f"\n{'='*60}")
|
||||
print("Run 2/2 — verifying determinism ...")
|
||||
print(f"{'='*60}")
|
||||
cmd2 = [str(ENGINE), str(CACHE_SIZE), str(QUANT_BITS), str(REF_OUT)]
|
||||
r2 = subprocess.run(cmd2, env=env, capture_output=True, text=True, cwd=str(HERE))
|
||||
print(r2.stdout)
|
||||
if r2.returncode != 0:
|
||||
print("STDERR:", r2.stderr, file=sys.stderr)
|
||||
sys.exit(r2.returncode)
|
||||
|
||||
if "Matching tokens: 12/12" in r2.stdout or f"Matching tokens: {MAX_NEW}/{MAX_NEW}" in r2.stdout:
|
||||
print("✓ Engine is DETERMINISTIC — same output on both runs!")
|
||||
else:
|
||||
m2 = re.search(r"Matching tokens: (\d+)/(\d+)", r2.stdout)
|
||||
if m2:
|
||||
print(f"⚠ Partial match: {m2.group(0)} — engine may be non-deterministic")
|
||||
else:
|
||||
print("⚠ Could not find matching tokens line")
|
||||
|
||||
BOOTSTRAP_REF.unlink(missing_ok=True)
|
||||
@@ -0,0 +1,5 @@
|
||||
"""colibrì — tiny engine, immense model."""
|
||||
|
||||
from colibri._version import __version__
|
||||
|
||||
__all__ = ["__version__"]
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Version accessor for the pip package.
|
||||
|
||||
The single source of truth is c/version.py (#394): coli --version and the
|
||||
GitHub Release workflow read it, so the pip metadata must read the SAME file
|
||||
instead of carrying a second literal that would drift on the first bump.
|
||||
|
||||
From a checkout (the supported install: `pip install -e .`) the file is read
|
||||
directly. From an installed wheel c/ is not on disk, so fall back to the
|
||||
package metadata that setuptools baked at build time from that same file.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
_ns = {}
|
||||
exec((Path(__file__).resolve().parent.parent / "c" / "version.py").read_text(), _ns)
|
||||
__version__ = _ns["__version__"]
|
||||
except OSError:
|
||||
from importlib.metadata import PackageNotFoundError, version
|
||||
|
||||
try:
|
||||
__version__ = version("colibri-engine")
|
||||
except PackageNotFoundError:
|
||||
__version__ = "0.0.0+unknown"
|
||||
@@ -0,0 +1,30 @@
|
||||
"""Entry point for `coli` when installed via pip.
|
||||
|
||||
Delegates to the original c/coli script which handles all subcommands.
|
||||
This wrapper exists so `pip install colibri-engine` creates a `coli` console
|
||||
script that works without the user having to add c/ to PATH manually.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import runpy
|
||||
|
||||
|
||||
def main():
|
||||
here = os.path.dirname(os.path.abspath(__file__))
|
||||
engine_dir = os.path.join(os.path.dirname(here), "c")
|
||||
coli_script = os.path.join(engine_dir, "coli")
|
||||
|
||||
if not os.path.exists(coli_script):
|
||||
sys.exit(
|
||||
"colibri engine directory not found.\n"
|
||||
"Install from source: git clone + pip install -e ."
|
||||
)
|
||||
|
||||
sys.path.insert(0, engine_dir)
|
||||
sys.argv[0] = coli_script
|
||||
runpy.run_path(coli_script, run_name="__main__")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+22
-3
@@ -41,7 +41,7 @@ Format: `VAR` — default — effect.
|
||||
| `PIPE` | `0` (off) | Overlap expert disk-load with matmul via I/O worker threads. Byte-identical output; reorders I/O. `PIPE=1` opts in. |
|
||||
| `PIPE_WORKERS` | `8` | Number of pthread loaders when `PIPE=1`, or the io-wq worker maximum per ring when `URING=1` (capped at 64). Tune to SSD queue depth and available cores. |
|
||||
| `URING` | `0` (off) | Linux-only queued expert I/O. `URING=1` implies `PIPE=1`, forces cold reads through io-wq (`IOSQE_ASYNC`), replaces blocking loader pthreads and spin waits with batched SQEs/CQEs, and batches `PILOT_REAL` loads on a separate ring. Use `DIRECT=1` for cold NVMe to avoid page-cache copy/readahead limits. Fails clearly if the kernel denies io_uring; incompatible with `COLI_MMAP=1`. |
|
||||
| `DIRECT` | `0` (off) | Use `O_DIRECT`/unbuffered reads for expert slabs. Helps sustained NVMe; keeps the zero-copy GPU path. |
|
||||
| `DIRECT` | `0` (off) | Use `O_DIRECT`/unbuffered reads for expert slabs. **Drive-dependent — measure it on your hardware.** On real NVMe with DRAM cache and headroom it is often a large win (measured +34% decode with `PIPE=1` on a Blackwell/Windows box, and 4.25→9.69 GB/s in iobench on a GB10); on QLC/DRAM-less drives or slow/virtualised disks it can be neutral to negative. Helps sustained NVMe; keeps the zero-copy GPU path. |
|
||||
| `COLI_NO_OMP_TUNE` | off | **Kill-switch** for the OpenMP hot-thread tuning (`OMP_WAIT_POLICY=active` spin + proc-bind). Set `=1` when the CPU is mostly waiting on the GPU (Metal) so spin doesn't steal the shared power budget. |
|
||||
| `COLI_NUMA` | auto in generated plans on multi-socket Linux; otherwise off | `COLI_NUMA=1` selectively interleaves large expert and dense slabs across NUMA nodes via `mbind` (raw syscall, no libnuma). Helps multi-socket hosts (+7–40% expert matmul); silent no-op on single-node or non-Linux. Explicit `COLI_NUMA=0` overrides the generated plan. |
|
||||
| `MLOCK` | `-1` (auto: on for macOS) | Wire the streamed expert cache into physical RAM (`mlock`) to dodge the memory compressor. `0` off, `1` force. |
|
||||
@@ -78,11 +78,21 @@ Format: `VAR` — default — effect.
|
||||
|
||||
---
|
||||
|
||||
## Dual-SSD streaming
|
||||
|
||||
| Variable | Default | Effect |
|
||||
|---|---|---|
|
||||
| `COLI_MODEL_DIRS` | unset | SPLIT the model across 2+ drives: a `;`/`,`-separated list of extra directories, each holding a **distinct** subset of the `.safetensors` shards (no duplication). Shards act as a search path — every shard is read from whichever drive holds it, so concurrent expert loads parallelise across drives and combined capacity is used. Scales to N drives. Metadata (config/tokenizer/`.coli_usage`) stays in the primary `COLI_MODEL` dir. Pairs well with `PIPE=1` (concurrent loaders) + `DIRECT=1`. Distinct from — and composable with — `COLI_MODEL_MIRROR`: the mirror is matched per-shard by basename against the merged (split) index, so a mirror dir may hold a copy of any subset of the split's shards. |
|
||||
| `COLI_MODEL_MIRROR` | unset | Path to a second, byte-identical (read-only) copy of the model on another drive; expert reads are split across both. Partial mirrors work (only the shards present are used). |
|
||||
| `COLI_DISK_WEIGHTS` | unset (startup bandwidth probe) | Split ratio `<primary>,<mirror>` (e.g. `1,1` for 50/50, `9,3` for a fast+slow pair). Unset = probe both drives with the engine's own access pattern at startup. |
|
||||
|
||||
Per-drive byte counts are reported in a `MIRROR:` stats line. Combine with `DIRECT=1` so the two copies never compete for page cache.
|
||||
|
||||
## CUDA (NVIDIA)
|
||||
|
||||
| Variable | Default | Effect |
|
||||
|---|---|---|
|
||||
| `COLI_CUDA` | off | Enable the CUDA backend. Requires a CUDA build. |
|
||||
| `COLI_CUDA` | off | Enable the CUDA backend. Requires a CUDA build. An explicit `COLI_CUDA=0` disables it **and suppresses the Windows bare-run auto-enable** (before this, Windows "CPU" runs with `COLI_CUDA=0` silently got a VRAM expert tier). The CLI flag `--gpu none` is the canonical hard off-switch on every platform. |
|
||||
| `COLI_GPU` / `COLI_GPUS` | unset | Device selection (`auto`, `none`, or a list like `0,1`). Requires `COLI_CUDA=1`. |
|
||||
| `CUDA_DENSE` | `0` | Place dense (non-expert) matmuls on the GPU. |
|
||||
| `CUDA_EXPERT_GB` | `0` | VRAM budget (GB) for caching experts on the GPU. |
|
||||
@@ -93,7 +103,7 @@ Format: `VAR` — default — effect.
|
||||
| `COLI_CUDA_PIPE` | `0` (off) | `1` engages the multi-step attention pipeline; `2` enables the pipe2 path. |
|
||||
| `COLI_CUDA_PIPE_SHARD` | off | `=1` runs the multi-device P2P head-shard attention path (opt-in for NVLink topologies; serializes ~95 MB/layer over a star PCIe topology). |
|
||||
| `COLI_CUDA_PIPE_S_MIN` | `1` single-GPU, `8` multi-GPU | Minimum prefill batch S to engage the pipe2 CUDA path. |
|
||||
| `COLI_CUDA_MTP` | `0` (off) | `=1` opts into MTP speculation under CUDA (off by default: cold streaming experts run on CPU where the fused-pair/IDOT kernels diverge in FP order, collapsing draft acceptance, #163/#292). |
|
||||
| `COLI_CUDA_MTP` | `0` (off) | `=1` opts into MTP speculation under CUDA (off by default: cold streaming experts run on CPU where the fused-pair/IDOT kernels diverge in FP order, collapsing draft acceptance, #163/#292 — though #467 measured acceptance holding at 49% on sm_120). When set explicitly, the resource planner skips its `DRAFT=0` export so the engine's auto path can engage draft=3 — no need to also set `DRAFT`. Note the measured trade-off (#467): at ~85% hit the widened S=4 expert union costs more than speculation saves (−32%); the opt-in pays only near-full residency (~99% hit). |
|
||||
| `COLI_CUDA_ASYNC` | on | `=0` forces synchronous `cudaMemcpy` instead of async + pinned host staging. |
|
||||
| `COLI_CUDA_DUAL_PROJ` | on | `=0` issues gate+up as two separate launches instead of one fused `grouped_hidden_w4_dual`. |
|
||||
| `COLI_CUDA_W4_PACKED` | on | `=0` disables the grouped packed-int4 path. |
|
||||
@@ -105,6 +115,15 @@ Format: `VAR` — default — effect.
|
||||
| `COLI_CUDA_SHARED_W4A16_MIN_ROWS` | `32` | Min row count to engage the shared-MLP W4A16 kernel. |
|
||||
| `COLI_METAL_UNTRACKED` | off (Metal only) | `=1` sets `MTLResourceHazardTrackingModeUntracked` on Metal buffers (reduces hazard-tracking overhead). |
|
||||
|
||||
> **Windows note.** On Windows, a bare `coli chat` / `coli run` / `coli serve`
|
||||
> (no `--gpu`/`--vram`/`--auto-tier`) **auto-enables the GPU** when it detects a
|
||||
> CUDA build (`coli_cuda.dll` next to the engine) and at least one GPU via
|
||||
> `nvidia-smi`. The expert-tier VRAM budget is then sized automatically from the
|
||||
> card's free VRAM (same computation as `--auto-tier`). If `nvidia-smi` is not on
|
||||
> `PATH` the run falls back to CPU with a warning — pass `--vram N` (or add
|
||||
> `nvidia-smi` to `PATH`) to enable CUDA in that case. `--gpu none` forces
|
||||
> CPU-only. (Linux/macOS behaviour is unchanged: pass a flag to enable CUDA.)
|
||||
|
||||
---
|
||||
|
||||
## Advanced / experimental / debug
|
||||
|
||||
@@ -83,3 +83,27 @@ this implementation adds: forced spans as a **draft source verified in the targe
|
||||
own forward** (lossless even under a wrong grammar, composes with MTP/n-gram in one
|
||||
union batch), deployed where the win is denominated in expert I/O rather than
|
||||
forward passes.
|
||||
|
||||
## Server usage: `response_format` (OpenAI API)
|
||||
|
||||
The gateway (`openai_server.py`) turns `response_format` into a **per-request**
|
||||
grammar carried to the engine over the `SUBMIT` protocol (optional 7th header
|
||||
field, `gbytes`; 6-field headers from older clients remain valid — the field is
|
||||
additive and back-compatible in both directions):
|
||||
|
||||
```jsonc
|
||||
{"response_format": {"type": "json_object"}} // generic JSON grammar
|
||||
{"response_format": {"type": "json_schema",
|
||||
"json_schema": {"schema": { ... }}}} // compiled by schema_gbnf.h
|
||||
{"response_format": {"type": "gbnf", "grammar": "root ::= ..."}} // raw GBNF (extension)
|
||||
```
|
||||
|
||||
Semantics are identical to `GRAMMAR=`/`SCHEMA=`: a **draft source, never a
|
||||
sampling constraint**. A schema outside the supported subset, or malformed GBNF,
|
||||
costs the speedup — never the request, never the output. Drafting engages for
|
||||
greedy requests (`temperature: 0`); sampled requests run undrafted. Compile
|
||||
overhead is negligible: ~8 µs/request for a typical schema, ~18 µs at the
|
||||
32-level nesting cap (measured, M3 Max). Grammar payloads are capped at 1 MiB.
|
||||
As with MTP (#100), a drafted greedy run may differ from an undrafted one in
|
||||
near-tie tokens (the verify forward has a different batch shape); each output
|
||||
is a valid greedy stream of its own forward shapes.
|
||||
|
||||
Generated
+3
-3
@@ -20,11 +20,11 @@
|
||||
},
|
||||
"nixpkgs": {
|
||||
"locked": {
|
||||
"lastModified": 1784160687,
|
||||
"narHash": "sha256-iYL/bixrb6FlHFu/gIuBYzq6c6lM5AAXsXNSWXtIgQc=",
|
||||
"lastModified": 1784280462,
|
||||
"narHash": "sha256-DtoqIqM7VkR6NxAkcLpMwmi02USwWb3JdmNGLyhthc0=",
|
||||
"owner": "NixOS",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "4382ed2b7a6839d4280a9b386db49cbc5907414d",
|
||||
"rev": "293d6abedf0478e681a4dfcfcb35b30fc796a32f",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
|
||||
@@ -6,37 +6,49 @@
|
||||
flake-utils.url = "github:numtide/flake-utils";
|
||||
};
|
||||
|
||||
outputs = { self, nixpkgs, flake-utils }:
|
||||
flake-utils.lib.eachDefaultSystem (system:
|
||||
let
|
||||
pkgs = import nixpkgs { inherit system; };
|
||||
outputs = {
|
||||
self,
|
||||
nixpkgs,
|
||||
flake-utils,
|
||||
}:
|
||||
flake-utils.lib.eachDefaultSystem (
|
||||
system: let
|
||||
pkgs = import nixpkgs {inherit system;};
|
||||
|
||||
# Python with the packages needed by the offline converter tools
|
||||
pythonEnv = pkgs.python3.withPackages (ps: with ps; [
|
||||
torch
|
||||
safetensors
|
||||
huggingface-hub
|
||||
numpy
|
||||
tokenizers
|
||||
datasets
|
||||
]);
|
||||
pythonEnv = pkgs.python3.withPackages (
|
||||
ps:
|
||||
with ps; [
|
||||
torch
|
||||
safetensors
|
||||
huggingface-hub
|
||||
numpy
|
||||
tokenizers
|
||||
datasets
|
||||
]
|
||||
);
|
||||
|
||||
colibri = pkgs.stdenv.mkDerivation {
|
||||
pname = "colibri";
|
||||
version = "1.0";
|
||||
src = ./.;
|
||||
|
||||
# python3 is needed by checkPhase: `make test-c` shells out to
|
||||
# `python3 tools/run_tests.py` (see c/Makefile, PYTHON ?= python3).
|
||||
nativeBuildInputs = [ pkgs.makeWrapper pkgs.python3 ];
|
||||
nativeBuildInputs = with pkgs; [makeWrapper];
|
||||
|
||||
buildInputs = [
|
||||
pkgs.gcc
|
||||
pkgs.gmp
|
||||
buildInputs = with pkgs; [
|
||||
gcc
|
||||
gmp
|
||||
];
|
||||
|
||||
# python3 is needed by checkPhase: `make test-c` shells out to
|
||||
# `python3 tools/run_tests.py` (see c/Makefile, PYTHON ?= python3).
|
||||
nativeCheckInputs = with pkgs; [python3];
|
||||
|
||||
# Use x86-64-v3 (AVX2) for a portable binary; override with ARCH=native for local builds
|
||||
ARCH = "x86-64-v3";
|
||||
ARCH =
|
||||
if pkgs.stdenv.hostPlatform.isx86_64
|
||||
then "x86-64-v3"
|
||||
else "native";
|
||||
|
||||
buildPhase = ''
|
||||
runHook preBuild
|
||||
@@ -56,7 +68,8 @@
|
||||
cp c/glm $out/lib/colibri/glm
|
||||
cp c/coli $out/lib/colibri/coli
|
||||
chmod +x $out/lib/colibri/coli
|
||||
cp c/openai_server.py c/resource_plan.py c/doctor.py $out/lib/colibri/
|
||||
cp c/openai_server.py c/resource_plan.py c/doctor.py c/version.py \
|
||||
$out/lib/colibri/
|
||||
cp -r c/tools/* $out/lib/colibri/tools/
|
||||
|
||||
# $out/bin holds the user-facing entry points.
|
||||
@@ -86,12 +99,11 @@
|
||||
description = "Run GLM-5.2 (744B MoE) on a consumer machine with ~25 GB RAM";
|
||||
homepage = "https://github.com/JustVugg/colibri";
|
||||
license = licenses.asl20;
|
||||
platforms = platforms.linux;
|
||||
mainProgram = "glm";
|
||||
platforms = with platforms; linux ++ darwin;
|
||||
mainProgram = "coli";
|
||||
};
|
||||
};
|
||||
in
|
||||
rec {
|
||||
in {
|
||||
packages = {
|
||||
default = colibri;
|
||||
inherit colibri;
|
||||
@@ -100,23 +112,25 @@
|
||||
apps = {
|
||||
default = {
|
||||
type = "app";
|
||||
program = "${colibri}/bin/glm";
|
||||
program = pkgs.lib.getExe colibri;
|
||||
};
|
||||
coli = {
|
||||
glm = {
|
||||
type = "app";
|
||||
program = "${colibri}/bin/coli";
|
||||
program = "${colibri}/share/colibri/glm";
|
||||
};
|
||||
};
|
||||
|
||||
devShells.default = pkgs.mkShell {
|
||||
inputsFrom = [ colibri ];
|
||||
formatter = pkgs.alejandra;
|
||||
|
||||
packages = [
|
||||
devShells.default = pkgs.mkShell {
|
||||
inputsFrom = [colibri];
|
||||
|
||||
packages = with pkgs; [
|
||||
pythonEnv
|
||||
pkgs.gcc
|
||||
pkgs.gnumake
|
||||
pkgs.clang-tools # clangd / clang-tidy for IDE support
|
||||
pkgs.pkg-config
|
||||
gcc
|
||||
gnumake
|
||||
clang-tools # clangd / clang-tidy for IDE support
|
||||
pkg-config
|
||||
];
|
||||
|
||||
shellHook = ''
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=68.0"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "colibri-engine"
|
||||
dynamic = ["version"]
|
||||
description = "Tiny engine, immense model — run GLM-5.2 (744B MoE) locally"
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
requires-python = ">=3.10"
|
||||
authors = [
|
||||
{name = "JustVugg"},
|
||||
]
|
||||
classifiers = [
|
||||
"Development Status :: 4 - Beta",
|
||||
"Environment :: Console",
|
||||
"Intended Audience :: Science/Research",
|
||||
"Operating System :: POSIX :: Linux",
|
||||
"Operating System :: MacOS",
|
||||
"Operating System :: Microsoft :: Windows",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
convert = [
|
||||
"numpy",
|
||||
"huggingface_hub",
|
||||
]
|
||||
oracle = [
|
||||
"torch>=2.0",
|
||||
"transformers>=4.40",
|
||||
"safetensors",
|
||||
]
|
||||
bench = [
|
||||
"tokenizers",
|
||||
"datasets",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
coli = "colibri.cli:main"
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/JustVugg/colibri"
|
||||
Issues = "https://github.com/JustVugg/colibri/issues"
|
||||
|
||||
[tool.setuptools.dynamic]
|
||||
version = {attr = "colibri._version.__version__"}
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["."]
|
||||
include = ["colibri*"]
|
||||
+64
-57
@@ -9,6 +9,7 @@ import {
|
||||
Database,
|
||||
Feather,
|
||||
Gauge,
|
||||
Globe,
|
||||
HardDrive,
|
||||
KeyRound,
|
||||
Layers,
|
||||
@@ -34,6 +35,7 @@ import { Brain } from "./Brain"
|
||||
import { Profiling } from "./Profiling"
|
||||
import { persistPublicSettings, stored } from "@/lib/storage"
|
||||
import { cn } from "@/lib/utils"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
const message = (role: ChatMessage["role"], content: string): ChatMessage => {
|
||||
let id: string
|
||||
@@ -42,15 +44,12 @@ const message = (role: ChatMessage["role"], content: string): ChatMessage => {
|
||||
}
|
||||
|
||||
export default function App() {
|
||||
// When the page is served by the engine itself (coli web), same-origin is the
|
||||
// right default: no CORS, no manual endpoint editing. The Vite dev server
|
||||
// (port 5173) keeps the classic default.
|
||||
const { t, locale, setLocale, locales } = useLocale()
|
||||
|
||||
const servedByEngine = typeof window !== "undefined" && window.location.port !== "5173" && window.location.protocol.startsWith("http")
|
||||
const defaultBase = servedByEngine ? `${window.location.origin}/v1` : "http://127.0.0.1:8000/v1"
|
||||
const [baseUrl, setBaseUrl] = useState(() => {
|
||||
const saved = stored(localStorage, "colibri.baseUrl", defaultBase)
|
||||
// migrate: a stored FACTORY default pointing at another origin would trip CORS
|
||||
// when the page is engine-served — upgrade it to same-origin once.
|
||||
if (servedByEngine && saved === "http://127.0.0.1:8000/v1" && defaultBase !== saved) return defaultBase
|
||||
return saved
|
||||
})
|
||||
@@ -120,7 +119,7 @@ export default function App() {
|
||||
const result = await getHealth(baseUrl, apiKey)
|
||||
if (!disposed) { setHealth(result); setHealthError("") }
|
||||
} catch (cause) {
|
||||
if (!disposed) setHealthError(cause instanceof Error ? cause.message : "Runtime metrics unavailable")
|
||||
if (!disposed) setHealthError(cause instanceof Error ? cause.message : "status.runtimeUnavailable")
|
||||
}
|
||||
}
|
||||
const timer = window.setInterval(() => void poll(), 5000)
|
||||
@@ -155,13 +154,13 @@ export default function App() {
|
||||
} catch (cause) {
|
||||
if (!controller.signal.aborted) {
|
||||
setHealth(null)
|
||||
setHealthError(cause instanceof Error ? cause.message : "Runtime metrics unavailable")
|
||||
setHealthError(cause instanceof Error ? cause.message : "status.runtimeUnavailable")
|
||||
}
|
||||
}
|
||||
} catch (cause) {
|
||||
if (controller.signal.aborted) return
|
||||
setConnected(false)
|
||||
setError(cause instanceof Error ? cause.message : "Could not reach the server.")
|
||||
setError(cause instanceof Error ? cause.message : "status.serverError")
|
||||
} finally {
|
||||
if (probeRef.current === controller) { probeRef.current = null; setConnecting(false) }
|
||||
}
|
||||
@@ -227,7 +226,7 @@ export default function App() {
|
||||
if (controller.signal.aborted) {
|
||||
updateMessages((current) => current.filter((item) => item.id !== assistant.id || item.content))
|
||||
} else {
|
||||
setError(cause instanceof Error ? cause.message : "Generation failed.")
|
||||
setError(cause instanceof Error ? cause.message : "status.generationFailed")
|
||||
updateMessages((current) => current.filter((item) => item.id !== assistant.id || item.content))
|
||||
}
|
||||
} finally {
|
||||
@@ -241,22 +240,22 @@ export default function App() {
|
||||
<aside className="sidebar">
|
||||
<div className="brand-row">
|
||||
<div className="brand-mark"><Feather className="size-5" /></div>
|
||||
<div><h1>colibrì</h1><p>local giant, tiny footprint</p></div>
|
||||
<div><h1>colibrì</h1><p>{t("brand.tagline")}</p></div>
|
||||
</div>
|
||||
|
||||
<section className="side-section">
|
||||
<div className="section-title"><Link2 className="size-3.5" /> Connection</div>
|
||||
<label>API endpoint<Input value={baseUrl} onChange={(event) => setBaseUrl(event.target.value)} /></label>
|
||||
<label>API key<div className="relative"><KeyRound className="field-icon" /><Input className="pl-9" type="password" value={apiKey} placeholder="optional" onChange={(event) => setApiKey(event.target.value)} /></div><span className="field-help">Kept in memory only · sent to this endpoint</span></label>
|
||||
<div className="section-title"><Link2 className="size-3.5" /> {t("sidebar.connection")}</div>
|
||||
<label>{t("sidebar.endpoint")}<Input value={baseUrl} onChange={(event) => setBaseUrl(event.target.value)} /></label>
|
||||
<label>{t("sidebar.apiKey")}<div className="relative"><KeyRound className="field-icon" /><Input className="pl-9" type="password" value={apiKey} placeholder={t("sidebar.apiKeyPlaceholder")} onChange={(event) => setApiKey(event.target.value)} /></div><span className="field-help">{t("sidebar.apiKeyHelp")}</span></label>
|
||||
<Button type="button" variant="secondary" onClick={connect} disabled={connecting}>
|
||||
{connecting ? <LoaderCircle className="size-4 animate-spin" /> : <RefreshCw className="size-4" />}
|
||||
Probe server
|
||||
{t("sidebar.probe")}
|
||||
</Button>
|
||||
<div className={cn("connection-state", connected && "connected")} aria-live="polite"><span />{connected ? "Engine reachable" : "Not connected"}</div>
|
||||
<div className={cn("connection-state", connected && "connected")} aria-live="polite"><span />{connected ? t("status.connected") : t("status.notConnected")}</div>
|
||||
</section>
|
||||
|
||||
<section className="side-section runtime-section" aria-live="polite">
|
||||
<div className="section-title"><Activity className="size-3.5" /> Runtime</div>
|
||||
<div className="section-title"><Activity className="size-3.5" /> {t("sidebar.runtime")}</div>
|
||||
{health?.hwinfo ? <div className="hw-panel">
|
||||
{health.hwinfo.cpu ? <div className="hw-row"><Cpu className="size-3.5" /><span>{health.hwinfo.cpu}</span></div> : null}
|
||||
{health.hwinfo.gpus > 0 ? <div className="hw-row"><MonitorDot className="size-3.5" /><span>{health.hwinfo.gpus}× GPU<small>{health.hwinfo.vram_total_gb.toFixed(0)} GB VRAM</small></span></div> : null}
|
||||
@@ -265,66 +264,74 @@ export default function App() {
|
||||
</div> : null}
|
||||
{health?.scheduler ? <>
|
||||
<div className="runtime-grid">
|
||||
<div><span>Active</span><strong>{active}<small> / {capacity}</small></strong></div>
|
||||
<div><span>Queued</span><strong>{health.scheduler.queued}<small> / {health.scheduler.max_queue}</small></strong></div>
|
||||
<div><span>Completed</span><strong>{health.scheduler.completed}</strong></div>
|
||||
<div><span>Failures</span><strong>{failures}</strong></div>
|
||||
<div><span>{t("dashboard.active")}</span><strong>{active}<small> / {capacity}</small></strong></div>
|
||||
<div><span>{t("dashboard.queued")}</span><strong>{health.scheduler.queued}<small> / {health.scheduler.max_queue}</small></strong></div>
|
||||
<div><span>{t("dashboard.completed")}</span><strong>{health.scheduler.completed}</strong></div>
|
||||
<div><span>{t("dashboard.failures")}</span><strong>{failures}</strong></div>
|
||||
</div>
|
||||
{health.tiers ? (() => {
|
||||
const t = health.tiers
|
||||
const total = Math.max(t.vram + t.ram + t.disk, 1)
|
||||
const ti = health.tiers
|
||||
const total = Math.max(ti.vram + ti.ram + ti.disk, 1)
|
||||
return <div className="tier-panel">
|
||||
<div className="tier-bar" role="img" aria-label={`Experts: ${t.vram} VRAM, ${t.ram} RAM, ${t.disk} disk`}>
|
||||
<span className="tier-vram" style={{ width: `${(100 * t.vram) / total}%` }} />
|
||||
<span className="tier-ram" style={{ width: `${(100 * t.ram) / total}%` }} />
|
||||
<span className="tier-disk" style={{ width: `${(100 * t.disk) / total}%` }} />
|
||||
<div className="tier-bar" role="img" aria-label={t("tier.ariaLabel", { vram: ti.vram, ram: ti.ram, disk: ti.disk })}>
|
||||
<span className="tier-vram" style={{ width: `${(100 * ti.vram) / total}%` }} />
|
||||
<span className="tier-ram" style={{ width: `${(100 * ti.ram) / total}%` }} />
|
||||
<span className="tier-disk" style={{ width: `${(100 * ti.disk) / total}%` }} />
|
||||
</div>
|
||||
<div className="tier-legend">
|
||||
<span><i className="tier-vram" />VRAM <strong>{t.vram.toLocaleString()}</strong><small>{t.vram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-ram" />RAM <strong>{t.ram.toLocaleString()}</strong><small>{t.ram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-disk" />Disk <strong>{t.disk.toLocaleString()}</strong></span>
|
||||
<span><i className="tier-vram" />{t("tier.vram")} <strong>{ti.vram.toLocaleString()}</strong><small>{ti.vram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-ram" />{t("tier.ram")} <strong>{ti.ram.toLocaleString()}</strong><small>{ti.ram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-disk" />{t("tier.disk")} <strong>{ti.disk.toLocaleString()}</strong></span>
|
||||
</div>
|
||||
</div>
|
||||
})() : null}
|
||||
{totalTokens.prompt + totalTokens.completion > 0 ? <div className="session-stats">
|
||||
<span><Database className="size-3" /> Session: <strong>{totalTokens.prompt.toLocaleString()}</strong> prompt + <strong>{totalTokens.completion.toLocaleString()}</strong> completion</span>
|
||||
<span><Database className="size-3" /> {t("dashboard.session")} <strong>{totalTokens.prompt.toLocaleString()}</strong> {t("dashboard.prompt")} + <strong>{totalTokens.completion.toLocaleString()}</strong> {t("dashboard.completion")}</span>
|
||||
</div> : null}
|
||||
<div className="runtime-foot"><span className="runtime-dot" /> Scheduler online <code>{kvSlots} KV</code></div>
|
||||
</> : <p className="runtime-unavailable">{connected ? (healthError || "Runtime metrics unavailable") : "Probe the server to inspect runtime state."}</p>}
|
||||
<div className="runtime-foot"><span className="runtime-dot" /> {t("sidebar.schedulerOnline")} <code>{kvSlots} KV</code></div>
|
||||
</> : <p className="runtime-unavailable">{connected ? (healthError ? t(healthError) : t("status.runtimeUnavailable")) : t("sidebar.runtimeProbe")}</p>}
|
||||
</section>
|
||||
|
||||
<section className="side-section">
|
||||
<div className="section-title"><SlidersHorizontal className="size-3.5" /> Inference</div>
|
||||
<label>Model<select value={model} onChange={(event) => setModel(event.target.value)}>{models.length ? models.map((id) => <option key={id}>{id}</option>) : <option>{model}</option>}</select></label>
|
||||
{health?.kv_slots && health.kv_slots > 1 ? <label>KV session<select value={cacheSlot} onChange={(event) => setCacheSlot(Number(event.target.value))} disabled={loading}>
|
||||
{Array.from({ length: kvSlots }, (_, slot) => <option key={slot} value={slot}>Session {slot + 1}</option>)}
|
||||
</select><span className="field-help">Isolated context · conversation follows the selected slot</span></label> : null}
|
||||
<label><span className="label-line"><span>Temperature</span><code>{temperature.toFixed(1)}</code></span><input className="range" type="range" min="0" max="2" step="0.1" value={temperature} onChange={(event) => setTemperature(Number(event.target.value))} /></label>
|
||||
<label>Max output tokens<Input type="number" min={1} max={4096} value={maxTokens} onChange={(event) => { const value = Number(event.target.value); if (Number.isFinite(value)) setMaxTokens(Math.min(4096, Math.max(1, Math.round(value)))) }} /></label>
|
||||
<div className="section-title"><SlidersHorizontal className="size-3.5" /> {t("sidebar.inference")}</div>
|
||||
<label>{t("sidebar.model")}<select value={model} onChange={(event) => setModel(event.target.value)}>{models.length ? models.map((id) => <option key={id}>{id}</option>) : <option>{model}</option>}</select></label>
|
||||
{health?.kv_slots && health.kv_slots > 1 ? <label>{t("sidebar.kvSession")}<select value={cacheSlot} onChange={(event) => setCacheSlot(Number(event.target.value))} disabled={loading}>
|
||||
{Array.from({ length: kvSlots }, (_, slot) => <option key={slot} value={slot}>{t("sidebar.sessionLabel", { slot: slot + 1 })}</option>)}
|
||||
</select><span className="field-help">{t("sidebar.kvSessionHelp")}</span></label> : null}
|
||||
<label><span className="label-line"><span>{t("sidebar.temperature")}</span><code>{temperature.toFixed(1)}</code></span><input className="range" type="range" min="0" max="2" step="0.1" value={temperature} onChange={(event) => setTemperature(Number(event.target.value))} /></label>
|
||||
<label>{t("sidebar.maxTokens")}<Input type="number" min={1} max={4096} value={maxTokens} onChange={(event) => { const value = Number(event.target.value); if (Number.isFinite(value)) setMaxTokens(Math.min(4096, Math.max(1, Math.round(value)))) }} /></label>
|
||||
<button type="button" className={cn("toggle-row", thinking && "active")} aria-pressed={thinking} onClick={() => setThinking((value) => !value)}>
|
||||
<span><BrainCircuit className="size-4" /> Reasoning</span><i><b /></i>
|
||||
<span><BrainCircuit className="size-4" /> {t("sidebar.reasoning")}</span><i><b /></i>
|
||||
</button>
|
||||
</section>
|
||||
|
||||
<div className="sidebar-foot"><Cpu className="size-3.5" /><span>OpenAI-compatible transport</span></div>
|
||||
<div className="sidebar-foot">
|
||||
<div><Cpu className="size-3.5" /><span>{t("sidebar.transport")}</span></div>
|
||||
<div className="locale-switcher">
|
||||
<Globe className="size-3.5" />
|
||||
<select value={locale} onChange={(e) => setLocale(e.target.value)}>
|
||||
{locales.map((l) => <option key={l.code} value={l.code}>{l.label}</option>)}
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<main className="chat-panel">
|
||||
<header className="topbar">
|
||||
<div><span className="eyebrow">ACTIVE MODEL</span><strong>{model}</strong></div>
|
||||
<div><span className="eyebrow">{t("topbar.activeModel")}</span><strong>{model}</strong></div>
|
||||
<div className="view-tabs">
|
||||
<button className={view === "chat" ? "active" : ""} onClick={() => setView("chat")}><MessageSquareText className="size-3.5" /> Chat</button>
|
||||
<button className={view === "brain" ? "active" : ""} onClick={() => setView("brain")}><BrainCircuit className="size-3.5" /> Brain</button>
|
||||
<button className={view === "profiling" ? "active" : ""} onClick={() => setView("profiling")}><Gauge className="size-3.5" /> Profiling</button>
|
||||
<button className={view === "chat" ? "active" : ""} onClick={() => setView("chat")}><MessageSquareText className="size-3.5" /> {t("nav.chat")}</button>
|
||||
<button className={view === "brain" ? "active" : ""} onClick={() => setView("brain")}><BrainCircuit className="size-3.5" /> {t("nav.brain")}</button>
|
||||
<button className={view === "profiling" ? "active" : ""} onClick={() => setView("profiling")}><Gauge className="size-3.5" /> {t("nav.profiling")}</button>
|
||||
</div>
|
||||
<div className="top-actions">
|
||||
{loading && tokenCount > 0 ? <Badge className="badge-live"><Zap className="size-3 flash" /> {tokenCount} tokens</Badge> : null}
|
||||
{!loading && tokPerSec != null ? <Badge className="badge-speed"><Gauge className="size-3" /> {tokPerSec.toFixed(1)} tok/s</Badge> : null}
|
||||
{loading && tokenCount > 0 ? <Badge className="badge-live"><Zap className="size-3 flash" /> {t("topbar.tokens", { n: tokenCount })}</Badge> : null}
|
||||
{!loading && tokPerSec != null ? <Badge className="badge-speed"><Gauge className="size-3" /> {t("topbar.tokPerSec", { n: tokPerSec.toFixed(1) })}</Badge> : null}
|
||||
{!loading && ttft != null ? <Badge><Timer className="size-3" /> TTFT {(ttft/1000).toFixed(1)}s</Badge> : null}
|
||||
{!loading && lastRun?.usage ? <Badge><Layers className="size-3" /> {lastRun.usage.prompt_tokens}→{lastRun.usage.completion_tokens}</Badge> : null}
|
||||
{lastRun?.queueWaitMs != null ? <Badge><Clock className="size-3" /> queue {Math.round(lastRun.queueWaitMs)}ms</Badge> : null}
|
||||
<Badge><MonitorDot className="size-3" /> slot {cacheSlot + 1}</Badge>
|
||||
<Button variant="ghost" size="sm" onClick={() => { updateMessages([]); setTokPerSec(null); setTtft(null); setTokenCount(0); setTotalTokens({prompt:0,completion:0}) }} disabled={!messages.length || loading}><Trash2 className="size-3.5" /> Clear</Button>
|
||||
<Badge><MonitorDot className="size-3" /> {t("topbar.slot", { n: cacheSlot + 1 })}</Badge>
|
||||
<Button variant="ghost" size="sm" onClick={() => { updateMessages([]); setTokPerSec(null); setTtft(null); setTokenCount(0); setTotalTokens({prompt:0,completion:0}) }} disabled={!messages.length || loading}><Trash2 className="size-3.5" /> {t("topbar.clear")}</Button>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
@@ -335,11 +342,11 @@ export default function App() {
|
||||
{!messages.length ? (
|
||||
<div className="empty-state">
|
||||
<div className="orb"><Feather /></div>
|
||||
<span className="eyebrow">COLIBRÌ ENGINE</span>
|
||||
<h2>Ask the giant.<br /><em>Keep the machine yours.</em></h2>
|
||||
<p>Connect to a local colibrì server and stream responses directly from your hardware. Nothing leaves the endpoint you choose.</p>
|
||||
<span className="eyebrow">{t("hero.title")}</span>
|
||||
<h2>{t("hero.subtitle")}<br /><em>{t("hero.tagline")}</em></h2>
|
||||
<p>{t("hero.description")}</p>
|
||||
<div className="suggestions">
|
||||
{["Explain how expert routing works", "Write a small C benchmark", "Compare RAM and VRAM caching"].map((item) => <button key={item} onClick={() => setDraft(item)}>{item}<ArrowUp className="size-3.5 rotate-45" /></button>)}
|
||||
{[t("prompts.routing"), t("prompts.benchmark"), t("prompts.caching")].map((item) => <button key={item} onClick={() => setDraft(item)}>{item}<ArrowUp className="size-3.5 rotate-45" /></button>)}
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
@@ -347,7 +354,7 @@ export default function App() {
|
||||
{messages.map((item) => (
|
||||
<article key={item.id} className={cn("message", item.role)}>
|
||||
<div className="avatar">{item.role === "user" ? "Y" : <Feather className="size-4" />}</div>
|
||||
<div><div className="message-meta">{item.role === "user" ? "You" : "colibrì"}</div><div className="message-body">{item.content || <span className="typing" aria-label="Generating"><i /><i /><i /></span>}</div></div>
|
||||
<div><div className="message-meta">{item.role === "user" ? t("chat.you") : t("chat.colibri")}</div><div className="message-body">{item.content || <span className="typing" aria-label="Generating"><i /><i /><i /></span>}</div></div>
|
||||
</article>
|
||||
))}
|
||||
<div ref={bottomRef} />
|
||||
@@ -356,10 +363,10 @@ export default function App() {
|
||||
</div>
|
||||
|
||||
<div className="composer-wrap">
|
||||
{error && <div className="error-banner" role="alert">{error}</div>}
|
||||
{error && <div className="error-banner" role="alert">{t(error)}</div>}
|
||||
<div className="composer">
|
||||
<Textarea value={draft} onChange={(event) => setDraft(event.target.value)} placeholder="Message colibrì…" onKeyDown={(event) => { if (event.key === "Enter" && !event.shiftKey && !event.nativeEvent.isComposing) { event.preventDefault(); void send() } }} />
|
||||
<div className="composer-foot"><span><MessageSquareText className="size-3.5" /> Enter to send · Shift+Enter for newline</span>{loading ? <Button variant="destructive" size="icon" aria-label="Stop generation" onClick={() => abortRef.current?.abort()}><CircleStop className="size-4" /></Button> : <Button size="icon" aria-label="Send message" disabled={!canSend} onClick={() => void send()}><ArrowUp className="size-4" /></Button>}</div>
|
||||
<Textarea value={draft} onChange={(event) => setDraft(event.target.value)} placeholder={t("chat.placeholder")} onKeyDown={(event) => { if (event.key === "Enter" && !event.shiftKey && !event.nativeEvent.isComposing) { event.preventDefault(); void send() } }} />
|
||||
<div className="composer-foot"><span><MessageSquareText className="size-3.5" /> {t("chat.inputHint")}</span>{loading ? <Button variant="destructive" size="icon" aria-label={t("chat.stop")} onClick={() => abortRef.current?.abort()}><CircleStop className="size-4" /></Button> : <Button size="icon" aria-label={t("chat.send")} disabled={!canSend} onClick={() => void send()}><ArrowUp className="size-4" /></Button>}</div>
|
||||
</div>
|
||||
</div>
|
||||
</>}
|
||||
|
||||
+21
-22
@@ -2,27 +2,26 @@ import { useEffect, useRef, useState } from "react"
|
||||
import { BrainCircuit, Flame, Layers } from "lucide-react"
|
||||
|
||||
import { endpoint } from "@/lib/api"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
interface ExpertMap { rows: number; cols: number; map: string; hits: string; seq: number }
|
||||
interface AtlasEntry { affinity: Record<string, number>; entropy: number; top: string; label: string }
|
||||
|
||||
const TIER_NAME = ["Disk", "RAM", "VRAM"]
|
||||
const TIER_KEYS = ["tier.disk", "tier.ram", "tier.vram"] as const
|
||||
const TIER_RGB: [number, number, number][] = [[58, 71, 80], [90, 155, 216], [78, 214, 165]]
|
||||
|
||||
/* Layer-depth heuristic: what this region of the network tends to specialise in.
|
||||
* Honest framing — these are the depth roles observed across MoE interpretability
|
||||
* work, not per-expert ground truth (that needs co-activation analysis, #119). */
|
||||
function depthRole(row: number, rows: number, isMtp: boolean): string {
|
||||
if (isMtp) return "MTP head — drafts the next token for speculative decoding"
|
||||
function depthRoleKey(row: number, rows: number, isMtp: boolean): string {
|
||||
if (isMtp) return "brain.mtp"
|
||||
const f = row / Math.max(rows - 1, 1)
|
||||
if (f < 0.2) return "early layers — surface features: tokens, spelling, local syntax"
|
||||
if (f < 0.45) return "lower-middle — phrase structure, word relations, simple facts"
|
||||
if (f < 0.7) return "upper-middle — semantics, long-range context, reasoning steps"
|
||||
if (f < 0.9) return "late layers — planning the answer, style, coherence"
|
||||
return "final layers — output shaping: picks the actual next-token distribution"
|
||||
if (f < 0.2) return "brain.early"
|
||||
if (f < 0.45) return "brain.lowerMiddle"
|
||||
if (f < 0.7) return "brain.upperMiddle"
|
||||
if (f < 0.9) return "brain.late"
|
||||
return "brain.final"
|
||||
}
|
||||
|
||||
export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey: string; connected: boolean }) {
|
||||
const { t } = useLocale()
|
||||
const canvasRef = useRef<HTMLCanvasElement>(null)
|
||||
const wrapRef = useRef<HTMLDivElement>(null)
|
||||
const [wrapSize, setWrapSize] = useState({ w: 1200, h: 700 })
|
||||
@@ -142,18 +141,18 @@ export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey:
|
||||
return (
|
||||
<div className="brain-page">
|
||||
<div className="brain-head">
|
||||
<div className="section-title"><BrainCircuit className="size-4" /> Expert Cortex — {data ? `${data.rows} layers × ${data.cols} experts` : "waiting for engine"}</div>
|
||||
<div className="section-title"><BrainCircuit className="size-4" /> {t("brain.title")} — {data ? t("brain.layers", { rows: data.rows, cols: data.cols }) : t("brain.waiting")}</div>
|
||||
<div className="brain-legend">
|
||||
<span><i style={{ background: "#4ed6a5" }} /> VRAM {totals[2].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#5a9bd8" }} /> RAM {totals[1].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#3a4750" }} /> Disk {totals[0].toLocaleString()}</span>
|
||||
<span><Flame className="size-3" /> brightness = routing heat</span>
|
||||
<span className="brain-pulse-hint">⚡ white flash = routed this turn</span>
|
||||
<span><i style={{ background: "#4ed6a5" }} /> {t("tier.vram")} {totals[2].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#5a9bd8" }} /> {t("tier.ram")} {totals[1].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#3a4750" }} /> {t("tier.disk")} {totals[0].toLocaleString()}</span>
|
||||
<span><Flame className="size-3" /> {t("brain.brightnessHint")}</span>
|
||||
<span className="brain-pulse-hint">{t("brain.flashHint")}</span>
|
||||
</div>
|
||||
</div>
|
||||
<div className="brain-canvas-wrap" ref={wrapRef}>
|
||||
<canvas ref={canvasRef} onMouseMove={onMove} onMouseLeave={() => setTip(null)} />
|
||||
{!connected && <p className="runtime-unavailable">Connect to the engine to see the cortex.</p>}
|
||||
{!connected && <p className="runtime-unavailable">{t("brain.connectHint")}</p>}
|
||||
</div>
|
||||
{tip && data && (() => {
|
||||
const isMtp = tip.row === data.rows - 1
|
||||
@@ -162,16 +161,16 @@ export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey:
|
||||
return (
|
||||
<div className="brain-tip" style={{ left: tip.x + 14, top: tip.y + 14 }}>
|
||||
<div className="brain-tip-title"><Layers className="size-3" /> Layer {realLayer}{isMtp ? " (MTP)" : ""} · Expert {tip.col}</div>
|
||||
<div>Tier: <strong style={{ color: ["#8b9aa3", "#5a9bd8", "#4ed6a5"][tip.tier] }}>{TIER_NAME[tip.tier]}</strong></div>
|
||||
<div>Heat: <strong>{tip.heat === 0 ? "never routed" : `~2^${tip.heat} selections`}</strong></div>
|
||||
<div>Tier: <strong style={{ color: ["#8b9aa3", "#5a9bd8", "#4ed6a5"][tip.tier] }}>{t(TIER_KEYS[tip.tier])}</strong></div>
|
||||
<div>Heat: <strong>{tip.heat === 0 ? t("brain.neverRouted") : t("brain.selections", { heat: tip.heat })}</strong></div>
|
||||
{entry ? <>
|
||||
<div className={entry.label.startsWith("specialist") ? "brain-tip-spec" : undefined}>
|
||||
{entry.label.startsWith("specialist") ? `⭐ Specialist: ${entry.top}` : "Generalist"}
|
||||
{entry.label.startsWith("specialist") ? t("brain.specialist", { top: entry.top }) : t("brain.generalist")}
|
||||
<small> (entropy {entry.entropy})</small>
|
||||
</div>
|
||||
<div className="brain-tip-aff">{Object.entries(entry.affinity).sort((a, b) => b[1] - a[1]).slice(0, 3)
|
||||
.map(([c, p]) => `${c} ${Math.round(p * 100)}%`).join(" · ")}</div>
|
||||
</> : <div className="brain-tip-role">{depthRole(tip.row, data.rows, isMtp)}</div>}
|
||||
</> : <div className="brain-tip-role">{t(depthRoleKey(tip.row, data.rows, isMtp))}</div>}
|
||||
</div>
|
||||
)
|
||||
})()}
|
||||
|
||||
@@ -1,4 +1,18 @@
|
||||
import { Component, type ReactNode } from "react"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
function ErrorFallback({ error, onRetry }: { error: Error; onRetry: () => void }) {
|
||||
const { t } = useLocale()
|
||||
return (
|
||||
<div style={{ padding: "2rem", fontFamily: "ui-monospace, monospace", color: "#e5e7eb", background: "#0b0f10", minHeight: "100vh" }}>
|
||||
<h2 style={{ color: "#4ed6a5" }}>{t("error.title")}</h2>
|
||||
<p style={{ color: "#9ca3af" }}>{t("error.hint")}</p>
|
||||
<pre style={{ whiteSpace: "pre-wrap", color: "#f87171" }}>{String(error)}</pre>
|
||||
<button onClick={onRetry} style={{ marginTop: "1rem", padding: "0.5rem 1rem", background: "#1f2937", color: "#e5e7eb", border: "1px solid #374151", borderRadius: 8, cursor: "pointer" }}>{t("error.retry")}</button>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
interface State { error: Error | null; stack: string }
|
||||
export class ErrorBoundary extends Component<{ children: ReactNode }, State> {
|
||||
state: State = { error: null, stack: "" }
|
||||
@@ -9,11 +23,6 @@ export class ErrorBoundary extends Component<{ children: ReactNode }, State> {
|
||||
}
|
||||
render() {
|
||||
if (!this.state.error) return this.props.children
|
||||
return <div style={{ padding: "2rem", fontFamily: "ui-monospace, monospace", color: "#e5e7eb", background: "#0b0f10", minHeight: "100vh" }}>
|
||||
<h2 style={{ color: "#4ed6a5" }}>colibrì UI hit an error</h2>
|
||||
<p style={{ color: "#9ca3af" }}>The engine is unaffected. Try refreshing.</p>
|
||||
<pre style={{ whiteSpace: "pre-wrap", color: "#f87171" }}>{String(this.state.error)}</pre>
|
||||
<button onClick={() => this.setState({ error: null, stack: "" })} style={{ marginTop: "1rem", padding: "0.5rem 1rem", background: "#1f2937", color: "#e5e7eb", border: "1px solid #374151", borderRadius: 8, cursor: "pointer" }}>Retry</button>
|
||||
</div>
|
||||
return <ErrorFallback error={this.state.error} onRetry={() => this.setState({ error: null, stack: "" })} />
|
||||
}
|
||||
}
|
||||
|
||||
+26
-32
@@ -2,19 +2,14 @@ import { useEffect, useState } from "react"
|
||||
import { Activity, Gauge, HardDrive, Timer } from "lucide-react"
|
||||
|
||||
import { getProfile, type ProfileTurn } from "@/lib/api"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
/* Wall-time phases stacked per turn. The order is the palette's CVD-safe slot
|
||||
* order (validated as a set on this surface) — identity never leans on colour
|
||||
* alone: segments keep 2px gaps, the legend is always shown and the table
|
||||
* carries the exact numbers. Disk *service* time is reported separately: it
|
||||
* runs on I/O threads overlapped with compute, so only the stall the compute
|
||||
* thread actually felt (I/O wait) belongs inside the wall-time stack. */
|
||||
const PHASES = [
|
||||
{ key: "expert_wait_s", name: "I/O wait", color: "#3987e5" },
|
||||
{ key: "expert_matmul_s", name: "Expert matmul", color: "#199e70" },
|
||||
{ key: "attention_s", name: "Attention", color: "#c98500" },
|
||||
{ key: "lm_head_s", name: "LM head", color: "#008300" },
|
||||
{ key: "other_s", name: "Other", color: "#9085e9" },
|
||||
{ key: "expert_wait_s", i18n: "profile.ioWait", color: "#3987e5" },
|
||||
{ key: "expert_matmul_s", i18n: "profile.expertMatmul", color: "#199e70" },
|
||||
{ key: "attention_s", i18n: "profile.attention", color: "#c98500" },
|
||||
{ key: "lm_head_s", i18n: "profile.lmHead", color: "#008300" },
|
||||
{ key: "other_s", i18n: "profile.other", color: "#9085e9" },
|
||||
] as const
|
||||
|
||||
interface Turn extends ProfileTurn { other_s: number; toks: number }
|
||||
@@ -28,8 +23,9 @@ const derive = (turn: ProfileTurn): Turn => ({
|
||||
const seconds = (value: number) => (value >= 10 ? value.toFixed(1) : value.toFixed(2)) + "s"
|
||||
|
||||
function ShareBar({ label, turns }: { label: string; turns: Turn[] }) {
|
||||
const { t } = useLocale()
|
||||
const total = turns.reduce((sum, turn) => sum + turn.wall_s, 0)
|
||||
const parts = PHASES.map((phase) => ({ ...phase, value: turns.reduce((sum, turn) => sum + turn[phase.key], 0) }))
|
||||
const parts = PHASES.map((phase) => ({ ...phase, name: t(phase.i18n), value: turns.reduce((sum, turn) => sum + turn[phase.key], 0) }))
|
||||
return (
|
||||
<div className="prof-share">
|
||||
<div className="prof-share-head"><span>{label}</span><code>{seconds(total)}</code></div>
|
||||
@@ -47,9 +43,7 @@ function ShareBar({ label, turns }: { label: string; turns: Turn[] }) {
|
||||
)
|
||||
}
|
||||
|
||||
/* Column chart over the recent turns; oldest on the left. Stacked mode draws the
|
||||
* wall-time composition, plain mode a single series (no legend — the title names it). */
|
||||
function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacked: boolean; height: number; format: (turn: Turn) => string }) {
|
||||
function TurnColumns({ turns, stacked, height, format, footLabel, footLabelOne }: { turns: Turn[]; stacked: boolean; height: number; format: (turn: Turn) => string; footLabel: string; footLabelOne: string }) {
|
||||
const [hover, setHover] = useState<number | null>(null)
|
||||
const peak = Math.max(...turns.map((turn) => (stacked ? turn.wall_s : turn.toks)), 1e-9)
|
||||
const gap = 2
|
||||
@@ -71,11 +65,10 @@ function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacke
|
||||
return h > 0.1 ? <rect key={`${index}-${phase.key}`} x={x} y={y + 0.35} width={width} height={Math.max(h - 0.7, 0.35)} fill={phase.color} opacity={hover === null || hover === index ? 1 : 0.45} /> : null
|
||||
})
|
||||
})}
|
||||
{/* hit targets bigger than the marks */}
|
||||
{turns.map((_, index) => <rect key={index} x={index * (width + gap) - gap / 2} y="0" width={width + gap} height={height} fill="transparent" onMouseEnter={() => setHover(index)} />)}
|
||||
</svg>
|
||||
<div className="prof-plot-foot">
|
||||
<span>{turns.length > 1 ? `${turns.length} turns · oldest → newest` : "1 turn"}</span>
|
||||
<span>{turns.length > 1 ? footLabel : footLabelOne}</span>
|
||||
<code>{hover !== null && turns[hover] ? format(turns[hover]) : `peak ${stacked ? seconds(peak) : peak.toFixed(1) + " tok/s"}`}</code>
|
||||
</div>
|
||||
</div>
|
||||
@@ -83,6 +76,7 @@ function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacke
|
||||
}
|
||||
|
||||
export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey: string; connected: boolean }) {
|
||||
const { t } = useLocale()
|
||||
const [turns, setTurns] = useState<Turn[]>([])
|
||||
|
||||
useEffect(() => {
|
||||
@@ -107,42 +101,42 @@ export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; api
|
||||
return (
|
||||
<div className="prof-page">
|
||||
<div className="prof-head">
|
||||
<div className="section-title"><Gauge className="size-4" /> Profiling — where the engine spends each turn</div>
|
||||
<div className="section-title"><Gauge className="size-4" /> {t("profile.title")}</div>
|
||||
<div className="prof-legend">
|
||||
{PHASES.map((phase) => <span key={phase.key}><i style={{ background: phase.color }} />{phase.name}</span>)}
|
||||
{PHASES.map((phase) => <span key={phase.key}><i style={{ background: phase.color }} />{t(phase.i18n)}</span>)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{!latest ? (
|
||||
<p className="runtime-unavailable">{connected ? "No profiled turns yet — send a chat message and the breakdown appears here." : "Connect to the engine to collect per-turn timings."}</p>
|
||||
<p className="runtime-unavailable">{connected ? t("profile.empty") : t("profile.connectHint")}</p>
|
||||
) : (
|
||||
<>
|
||||
<div className="prof-tiles">
|
||||
<div><span><Gauge className="size-3" /> Last turn</span><strong>{latest.toks.toFixed(1)}</strong><small>tok/s</small></div>
|
||||
<div><span><Timer className="size-3" /> Wall time</span><strong>{seconds(latest.wall_s)}</strong><small>{latest.prompt_tokens} → {latest.completion_tokens} tokens</small></div>
|
||||
<div><span><Activity className="size-3" /> Batching</span><strong>{latest.forwards > 0 ? (latest.completion_tokens / latest.forwards).toFixed(2) : "—"}</strong><small>tokens / forward</small></div>
|
||||
<div><span><HardDrive className="size-3" /> Disk service</span><strong>{seconds(latest.expert_disk_s)}</strong><small>overlapped with compute</small></div>
|
||||
<div><span><Gauge className="size-3" /> {t("profile.lastTurn")}</span><strong>{latest.toks.toFixed(1)}</strong><small>tok/s</small></div>
|
||||
<div><span><Timer className="size-3" /> {t("profile.wallTime")}</span><strong>{seconds(latest.wall_s)}</strong><small>{latest.prompt_tokens} → {latest.completion_tokens} tokens</small></div>
|
||||
<div><span><Activity className="size-3" /> {t("profile.batching")}</span><strong>{latest.forwards > 0 ? (latest.completion_tokens / latest.forwards).toFixed(2) : "—"}</strong><small>{t("profile.tokensPerForward")}</small></div>
|
||||
<div><span><HardDrive className="size-3" /> {t("profile.diskService")}</span><strong>{seconds(latest.expert_disk_s)}</strong><small>{t("profile.overlapped")}</small></div>
|
||||
</div>
|
||||
|
||||
<div className="prof-shares">
|
||||
<ShareBar label="Last turn" turns={[latest]} />
|
||||
{turns.length > 1 ? <ShareBar label={`Window · last ${turns.length} turns`} turns={turns} /> : null}
|
||||
<ShareBar label={t("profile.lastTurn")} turns={[latest]} />
|
||||
{turns.length > 1 ? <ShareBar label={t("profile.window", { n: turns.length })} turns={turns} /> : null}
|
||||
</div>
|
||||
|
||||
<div className="prof-charts">
|
||||
<div className="prof-chart">
|
||||
<div className="prof-chart-title">Throughput per turn (tok/s)</div>
|
||||
<TurnColumns turns={recent} stacked={false} height={36} format={(turn) => `${turn.toks.toFixed(1)} tok/s · ${turn.completion_tokens} tokens`} />
|
||||
<div className="prof-chart-title">{t("profile.throughputTitle")}</div>
|
||||
<TurnColumns turns={recent} stacked={false} height={36} footLabel={t("profile.turnsLabel", { n: recent.length })} footLabelOne={t("profile.oneTurn")} format={(turn) => `${turn.toks.toFixed(1)} tok/s · ${turn.completion_tokens} tokens`} />
|
||||
</div>
|
||||
<div className="prof-chart">
|
||||
<div className="prof-chart-title">Turn wall time by phase (s)</div>
|
||||
<TurnColumns turns={recent} stacked height={36} format={(turn) => `${seconds(turn.wall_s)} · ${PHASES.map((phase) => `${phase.name} ${seconds(turn[phase.key])}`).join(" · ")}`} />
|
||||
<div className="prof-chart-title">{t("profile.phaseTitle")}</div>
|
||||
<TurnColumns turns={recent} stacked height={36} footLabel={t("profile.turnsLabel", { n: recent.length })} footLabelOne={t("profile.oneTurn")} format={(turn) => `${seconds(turn.wall_s)} · ${PHASES.map((phase) => `${t(phase.i18n)} ${seconds(turn[phase.key])}`).join(" · ")}`} />
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="prof-table-wrap">
|
||||
<table className="prof-table">
|
||||
<thead><tr><th>Turn</th><th>Tokens</th><th>tok/s</th><th>Wall</th>{PHASES.map((phase) => <th key={phase.key}><i style={{ background: phase.color }} />{phase.name}</th>)}<th>Disk service</th></tr></thead>
|
||||
<thead><tr><th>{t("profile.turnCol")}</th><th>{t("profile.tokensCol")}</th><th>tok/s</th><th>{t("profile.wallCol")}</th>{PHASES.map((phase) => <th key={phase.key}><i style={{ background: phase.color }} />{t(phase.i18n)}</th>)}<th>{t("profile.diskService")}</th></tr></thead>
|
||||
<tbody>
|
||||
{recent.slice().reverse().map((turn, index) => (
|
||||
<tr key={turns.length - index}>
|
||||
@@ -156,7 +150,7 @@ export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; api
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
{diskService > 0 ? <p className="prof-note">Disk service is time spent reading experts on I/O threads; it overlaps with compute, so only the <em>I/O wait</em> the compute thread felt counts inside the wall-time stack. With multiple KV sessions the shares describe the whole engine over the turn's window.</p> : null}
|
||||
{diskService > 0 ? <p className="prof-note">{t("profile.diskNote")}</p> : null}
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
const en: Record<string, string> = {
|
||||
// nav
|
||||
"nav.chat": "Chat",
|
||||
"nav.brain": "Brain",
|
||||
"nav.profiling": "Profiling",
|
||||
|
||||
// brand
|
||||
"brand.tagline": "local giant, tiny footprint",
|
||||
|
||||
// sidebar — connection
|
||||
"sidebar.connection": "Connection",
|
||||
"sidebar.endpoint": "API endpoint",
|
||||
"sidebar.apiKey": "API key",
|
||||
"sidebar.apiKeyPlaceholder": "optional",
|
||||
"sidebar.apiKeyHelp": "Kept in memory only · sent to this endpoint",
|
||||
"sidebar.probe": "Probe server",
|
||||
"status.connected": "Engine reachable",
|
||||
"status.notConnected": "Not connected",
|
||||
"status.runtimeUnavailable": "Runtime metrics unavailable",
|
||||
"status.serverError": "Could not reach the server.",
|
||||
"status.generationFailed": "Generation failed.",
|
||||
|
||||
// sidebar — runtime
|
||||
"sidebar.runtime": "Runtime",
|
||||
"sidebar.runtimeProbe": "Probe the server to inspect runtime state.",
|
||||
"sidebar.schedulerOnline": "Scheduler online",
|
||||
"dashboard.active": "Active",
|
||||
"dashboard.queued": "Queued",
|
||||
"dashboard.completed": "Completed",
|
||||
"dashboard.failures": "Failures",
|
||||
"dashboard.session": "Session:",
|
||||
"dashboard.prompt": "prompt",
|
||||
"dashboard.completion": "completion",
|
||||
|
||||
// sidebar — tiers
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "Disk",
|
||||
"tier.ariaLabel": "Experts: {{vram}} VRAM, {{ram}} RAM, {{disk}} disk",
|
||||
|
||||
// sidebar — inference
|
||||
"sidebar.inference": "Inference",
|
||||
"sidebar.model": "Model",
|
||||
"sidebar.kvSession": "KV session",
|
||||
"sidebar.kvSessionHelp": "Isolated context · conversation follows the selected slot",
|
||||
"sidebar.sessionLabel": "Session {{slot}}",
|
||||
"sidebar.temperature": "Temperature",
|
||||
"sidebar.maxTokens": "Max output tokens",
|
||||
"sidebar.reasoning": "Reasoning",
|
||||
"sidebar.transport": "OpenAI-compatible transport",
|
||||
|
||||
// top bar
|
||||
"topbar.activeModel": "ACTIVE MODEL",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "slot {{n}}",
|
||||
"topbar.clear": "Clear",
|
||||
|
||||
// hero / empty state
|
||||
"hero.title": "COLIBRÌ ENGINE",
|
||||
"hero.subtitle": "Ask the giant.",
|
||||
"hero.tagline": "Keep the machine yours.",
|
||||
"hero.description": "Connect to a local colibrì server and stream responses directly from your hardware. Nothing leaves the endpoint you choose.",
|
||||
"prompts.routing": "Explain how expert routing works",
|
||||
"prompts.benchmark": "Write a small C benchmark",
|
||||
"prompts.caching": "Compare RAM and VRAM caching",
|
||||
|
||||
// chat
|
||||
"chat.you": "You",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "Message colibrì…",
|
||||
"chat.inputHint": "Enter to send · Shift+Enter for newline",
|
||||
"chat.stop": "Stop generation",
|
||||
"chat.send": "Send message",
|
||||
|
||||
// brain
|
||||
"brain.title": "Expert Cortex",
|
||||
"brain.waiting": "waiting for engine",
|
||||
"brain.layers": "{{rows}} layers × {{cols}} experts",
|
||||
"brain.brightnessHint": "brightness = routing heat",
|
||||
"brain.flashHint": "⚡ white flash = routed this turn",
|
||||
"brain.connectHint": "Connect to the engine to see the cortex.",
|
||||
"brain.neverRouted": "never routed",
|
||||
"brain.selections": "~2^{{heat}} selections",
|
||||
"brain.specialist": "⭐ Specialist: {{top}}",
|
||||
"brain.generalist": "Generalist",
|
||||
"brain.mtp": "MTP head — drafts the next token for speculative decoding",
|
||||
"brain.early": "early layers — surface features: tokens, spelling, local syntax",
|
||||
"brain.lowerMiddle": "lower-middle — phrase structure, word relations, simple facts",
|
||||
"brain.upperMiddle": "upper-middle — semantics, long-range context, reasoning steps",
|
||||
"brain.late": "late layers — planning the answer, style, coherence",
|
||||
"brain.final": "final layers — output shaping: picks the actual next-token distribution",
|
||||
|
||||
// profiling
|
||||
"profile.title": "Profiling — where the engine spends each turn",
|
||||
"profile.ioWait": "I/O wait",
|
||||
"profile.expertMatmul": "Expert matmul",
|
||||
"profile.attention": "Attention",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "Other",
|
||||
"profile.empty": "No profiled turns yet — send a chat message and the breakdown appears here.",
|
||||
"profile.connectHint": "Connect to the engine to collect per-turn timings.",
|
||||
"profile.lastTurn": "Last turn",
|
||||
"profile.wallTime": "Wall time",
|
||||
"profile.batching": "Batching",
|
||||
"profile.tokensPerForward": "tokens / forward",
|
||||
"profile.diskService": "Disk service",
|
||||
"profile.overlapped": "overlapped with compute",
|
||||
"profile.window": "Window · last {{n}} turns",
|
||||
"profile.throughputTitle": "Throughput per turn (tok/s)",
|
||||
"profile.phaseTitle": "Turn wall time by phase (s)",
|
||||
"profile.turnCol": "Turn",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "Wall",
|
||||
"profile.turnsLabel": "{{n}} turns · oldest → newest",
|
||||
"profile.oneTurn": "1 turn",
|
||||
"profile.diskNote": "Disk service is time spent reading experts on I/O threads; it overlaps with compute, so only the I/O wait the compute thread felt counts inside the wall-time stack. With multiple KV sessions the shares describe the whole engine over the turn's window.",
|
||||
|
||||
// error boundary
|
||||
"error.title": "colibrì UI hit an error",
|
||||
"error.hint": "The engine is unaffected. Try refreshing.",
|
||||
"error.retry": "Retry",
|
||||
}
|
||||
|
||||
export default en
|
||||
@@ -0,0 +1,78 @@
|
||||
import { createContext, useContext, useState, useCallback, useMemo, type ReactNode } from "react"
|
||||
import { createElement } from "react"
|
||||
import en from "./en"
|
||||
import zhCN from "./zh-CN"
|
||||
import zhTW from "./zh-TW"
|
||||
import it from "./it"
|
||||
|
||||
const LOCALES = [
|
||||
{ code: "en", label: "English" },
|
||||
{ code: "zh-CN", label: "简体中文" },
|
||||
{ code: "zh-TW", label: "繁體中文" },
|
||||
{ code: "it", label: "Italiano" },
|
||||
] as const
|
||||
|
||||
const DICTS: Record<string, Record<string, string>> = {
|
||||
"en": en,
|
||||
"zh-CN": zhCN,
|
||||
"zh-TW": zhTW,
|
||||
"it": it,
|
||||
}
|
||||
|
||||
const STORAGE_KEY = "colibri-locale"
|
||||
|
||||
function detectLocale(): string {
|
||||
try {
|
||||
const saved = localStorage.getItem(STORAGE_KEY)
|
||||
if (saved && DICTS[saved]) return saved
|
||||
} catch {}
|
||||
const nav = navigator.language || ""
|
||||
if (DICTS[nav]) return nav
|
||||
const prefix = nav.split("-")[0]
|
||||
if (prefix === "zh") return nav.includes("TW") || nav.includes("Hant") ? "zh-TW" : "zh-CN"
|
||||
for (const { code } of LOCALES) if (code.startsWith(prefix)) return code
|
||||
return "en"
|
||||
}
|
||||
|
||||
function interpolate(template: string, vars?: Record<string, string | number>): string {
|
||||
if (!vars) return template
|
||||
return template.replace(/\{\{(\w+)\}\}/g, (_, key) => String(vars[key] ?? `{{${key}}}`))
|
||||
}
|
||||
|
||||
interface LocaleContext {
|
||||
locale: string
|
||||
setLocale: (code: string) => void
|
||||
t: (key: string, vars?: Record<string, string | number>) => string
|
||||
locales: readonly { code: string; label: string }[]
|
||||
}
|
||||
|
||||
const Ctx = createContext<LocaleContext>({
|
||||
locale: "en",
|
||||
setLocale: () => {},
|
||||
t: (key) => key,
|
||||
locales: LOCALES,
|
||||
})
|
||||
|
||||
export function LocaleProvider({ children }: { children: ReactNode }) {
|
||||
const [locale, setLocaleState] = useState(detectLocale)
|
||||
|
||||
const setLocale = useCallback((code: string) => {
|
||||
if (!DICTS[code]) return
|
||||
setLocaleState(code)
|
||||
try { localStorage.setItem(STORAGE_KEY, code) } catch {}
|
||||
}, [])
|
||||
|
||||
const t = useCallback((key: string, vars?: Record<string, string | number>) => {
|
||||
const dict = DICTS[locale] || en
|
||||
const template = dict[key] ?? en[key] ?? key
|
||||
return interpolate(template, vars)
|
||||
}, [locale])
|
||||
|
||||
const value = useMemo(() => ({ locale, setLocale, t, locales: LOCALES }), [locale, setLocale, t])
|
||||
|
||||
return createElement(Ctx.Provider, { value }, children)
|
||||
}
|
||||
|
||||
export function useLocale() {
|
||||
return useContext(Ctx)
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
const it: Record<string, string> = {
|
||||
"nav.chat": "Chat",
|
||||
"nav.brain": "Cervello",
|
||||
"nav.profiling": "Profiling",
|
||||
|
||||
"brand.tagline": "gigante locale, impronta minima",
|
||||
|
||||
"sidebar.connection": "Connessione",
|
||||
"sidebar.endpoint": "Endpoint API",
|
||||
"sidebar.apiKey": "Chiave API",
|
||||
"sidebar.apiKeyPlaceholder": "opzionale",
|
||||
"sidebar.apiKeyHelp": "Conservata solo in memoria · inviata a questo endpoint",
|
||||
"sidebar.probe": "Sonda il server",
|
||||
"status.connected": "Motore raggiungibile",
|
||||
"status.notConnected": "Non connesso",
|
||||
"status.runtimeUnavailable": "Metriche runtime non disponibili",
|
||||
"status.serverError": "Impossibile raggiungere il server.",
|
||||
"status.generationFailed": "Generazione fallita.",
|
||||
|
||||
"sidebar.runtime": "Runtime",
|
||||
"sidebar.runtimeProbe": "Sonda il server per ispezionare lo stato runtime.",
|
||||
"sidebar.schedulerOnline": "Scheduler online",
|
||||
"dashboard.active": "Attive",
|
||||
"dashboard.queued": "In coda",
|
||||
"dashboard.completed": "Completate",
|
||||
"dashboard.failures": "Fallite",
|
||||
"dashboard.session": "Sessione:",
|
||||
"dashboard.prompt": "prompt",
|
||||
"dashboard.completion": "completion",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "Disco",
|
||||
"tier.ariaLabel": "Expert: {{vram}} VRAM, {{ram}} RAM, {{disk}} disco",
|
||||
|
||||
"sidebar.inference": "Inferenza",
|
||||
"sidebar.model": "Modello",
|
||||
"sidebar.kvSession": "Sessione KV",
|
||||
"sidebar.kvSessionHelp": "Contesto isolato · la conversazione segue lo slot selezionato",
|
||||
"sidebar.sessionLabel": "Sessione {{slot}}",
|
||||
"sidebar.temperature": "Temperatura",
|
||||
"sidebar.maxTokens": "Token di output massimi",
|
||||
"sidebar.reasoning": "Ragionamento",
|
||||
"sidebar.transport": "Trasporto compatibile OpenAI",
|
||||
|
||||
"topbar.activeModel": "MODELLO ATTIVO",
|
||||
"topbar.tokens": "{{n}} token",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "slot {{n}}",
|
||||
"topbar.clear": "Pulisci",
|
||||
|
||||
"hero.title": "MOTORE COLIBRÌ",
|
||||
"hero.subtitle": "Interroga il gigante.",
|
||||
"hero.tagline": "La macchina resta tua.",
|
||||
"hero.description": "Connettiti a un server colibrì locale e ricevi le risposte in streaming direttamente dal tuo hardware. Nulla lascia l'endpoint che scegli.",
|
||||
"prompts.routing": "Spiega come funziona il routing degli expert",
|
||||
"prompts.benchmark": "Scrivi un piccolo benchmark in C",
|
||||
"prompts.caching": "Confronta il caching RAM e VRAM",
|
||||
|
||||
"chat.you": "Tu",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "Scrivi a colibrì…",
|
||||
"chat.inputHint": "Invio per inviare · Shift+Invio per andare a capo",
|
||||
"chat.stop": "Ferma la generazione",
|
||||
"chat.send": "Invia messaggio",
|
||||
|
||||
"brain.title": "Corteccia degli expert",
|
||||
"brain.waiting": "in attesa del motore",
|
||||
"brain.layers": "{{rows}} layer × {{cols}} expert",
|
||||
"brain.brightnessHint": "luminosità = calore di routing",
|
||||
"brain.flashHint": "⚡ flash bianco = instradato in questo turno",
|
||||
"brain.connectHint": "Connettiti al motore per vedere la corteccia.",
|
||||
"brain.neverRouted": "mai instradato",
|
||||
"brain.selections": "~2^{{heat}} selezioni",
|
||||
"brain.specialist": "⭐ Specialista: {{top}}",
|
||||
"brain.generalist": "Generalista",
|
||||
"brain.mtp": "Testa MTP — prepara il prossimo token per la decodifica speculativa",
|
||||
"brain.early": "layer iniziali — caratteristiche superficiali: token, ortografia, sintassi locale",
|
||||
"brain.lowerMiddle": "layer medio-bassi — struttura frasale, relazioni tra parole, fatti semplici",
|
||||
"brain.upperMiddle": "layer medio-alti — semantica, contesto a lungo raggio, passi di ragionamento",
|
||||
"brain.late": "layer avanzati — pianificazione della risposta, stile, coerenza",
|
||||
"brain.final": "layer finali — formazione dell'output: scelta della distribuzione next-token",
|
||||
|
||||
"profile.title": "Profiling — dove il motore spende ogni turno",
|
||||
"profile.ioWait": "Attesa I/O",
|
||||
"profile.expertMatmul": "Matmul expert",
|
||||
"profile.attention": "Attenzione",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "Altro",
|
||||
"profile.empty": "Nessun turno profilato — invia un messaggio e i dettagli appariranno qui.",
|
||||
"profile.connectHint": "Connettiti al motore per raccogliere i tempi per turno.",
|
||||
"profile.lastTurn": "Ultimo turno",
|
||||
"profile.wallTime": "Tempo totale",
|
||||
"profile.batching": "Batching",
|
||||
"profile.tokensPerForward": "token / forward",
|
||||
"profile.diskService": "Servizio disco",
|
||||
"profile.overlapped": "sovrapposto al calcolo",
|
||||
"profile.window": "Finestra · ultimi {{n}} turni",
|
||||
"profile.throughputTitle": "Throughput per turno (tok/s)",
|
||||
"profile.phaseTitle": "Tempo per turno per fase (s)",
|
||||
"profile.turnCol": "Turno",
|
||||
"profile.tokensCol": "Token",
|
||||
"profile.wallCol": "Totale",
|
||||
"profile.turnsLabel": "{{n}} turni · dal meno al più recente",
|
||||
"profile.oneTurn": "1 turno",
|
||||
"profile.diskNote": "Il servizio disco è il tempo speso a leggere gli expert sui thread I/O; si sovrappone al calcolo, quindi solo l'attesa I/O effettivamente percepita dal thread di calcolo conta nella ripartizione del tempo totale. Con più sessioni KV, le quote descrivono l'intero motore nella finestra del turno.",
|
||||
|
||||
"error.title": "L'interfaccia colibrì ha riscontrato un errore",
|
||||
"error.hint": "Il motore non è stato coinvolto. Prova a ricaricare la pagina.",
|
||||
"error.retry": "Riprova",
|
||||
}
|
||||
|
||||
export default it
|
||||
@@ -0,0 +1,113 @@
|
||||
const zhCN: Record<string, string> = {
|
||||
"nav.chat": "对话",
|
||||
"nav.brain": "大脑",
|
||||
"nav.profiling": "性能分析",
|
||||
|
||||
"brand.tagline": "本地巨人,极小足迹",
|
||||
|
||||
"sidebar.connection": "连接",
|
||||
"sidebar.endpoint": "API 端点",
|
||||
"sidebar.apiKey": "API 密钥",
|
||||
"sidebar.apiKeyPlaceholder": "可选",
|
||||
"sidebar.apiKeyHelp": "仅保存在内存中 · 发送到此端点",
|
||||
"sidebar.probe": "探测服务器",
|
||||
"status.connected": "引擎已连接",
|
||||
"status.notConnected": "未连接",
|
||||
"status.runtimeUnavailable": "运行时指标不可用",
|
||||
"status.serverError": "无法连接到服务器。",
|
||||
"status.generationFailed": "生成失败。",
|
||||
|
||||
"sidebar.runtime": "运行时",
|
||||
"sidebar.runtimeProbe": "探测服务器以查看运行时状态。",
|
||||
"sidebar.schedulerOnline": "调度器在线",
|
||||
"dashboard.active": "活跃",
|
||||
"dashboard.queued": "排队",
|
||||
"dashboard.completed": "已完成",
|
||||
"dashboard.failures": "失败",
|
||||
"dashboard.session": "会话:",
|
||||
"dashboard.prompt": "提示词",
|
||||
"dashboard.completion": "补全",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "磁盘",
|
||||
"tier.ariaLabel": "专家分布:{{vram}} VRAM、{{ram}} RAM、{{disk}} 磁盘",
|
||||
|
||||
"sidebar.inference": "推理",
|
||||
"sidebar.model": "模型",
|
||||
"sidebar.kvSession": "KV 会话",
|
||||
"sidebar.kvSessionHelp": "独立上下文 · 对话跟随所选槽位",
|
||||
"sidebar.sessionLabel": "会话 {{slot}}",
|
||||
"sidebar.temperature": "温度",
|
||||
"sidebar.maxTokens": "最大输出 token 数",
|
||||
"sidebar.reasoning": "推理模式",
|
||||
"sidebar.transport": "OpenAI 兼容协议",
|
||||
|
||||
"topbar.activeModel": "当前模型",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "槽位 {{n}}",
|
||||
"topbar.clear": "清空",
|
||||
|
||||
"hero.title": "COLIBRÌ 引擎",
|
||||
"hero.subtitle": "向巨人提问。",
|
||||
"hero.tagline": "让机器属于你。",
|
||||
"hero.description": "连接到本地 colibrì 服务器,直接从你的硬件流式获取响应。所有数据都留在你选择的端点内。",
|
||||
"prompts.routing": "解释专家路由是如何工作的",
|
||||
"prompts.benchmark": "写一个简单的 C 基准测试",
|
||||
"prompts.caching": "比较 RAM 和 VRAM 缓存",
|
||||
|
||||
"chat.you": "你",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "给 colibrì 发消息…",
|
||||
"chat.inputHint": "回车发送 · Shift+回车换行",
|
||||
"chat.stop": "停止生成",
|
||||
"chat.send": "发送消息",
|
||||
|
||||
"brain.title": "专家皮层",
|
||||
"brain.waiting": "等待引擎连接",
|
||||
"brain.layers": "{{rows}} 层 × {{cols}} 专家",
|
||||
"brain.brightnessHint": "亮度 = 路由热度",
|
||||
"brain.flashHint": "⚡ 白色闪烁 = 本轮被路由",
|
||||
"brain.connectHint": "连接引擎以查看皮层。",
|
||||
"brain.neverRouted": "从未被路由",
|
||||
"brain.selections": "约 2^{{heat}} 次选择",
|
||||
"brain.specialist": "⭐ 专精:{{top}}",
|
||||
"brain.generalist": "通用型",
|
||||
"brain.mtp": "MTP 头 — 为投机解码起草下一个 token",
|
||||
"brain.early": "早期层 — 表面特征:token、拼写、局部语法",
|
||||
"brain.lowerMiddle": "中低层 — 短语结构、词语关系、简单事实",
|
||||
"brain.upperMiddle": "中高层 — 语义、长距离上下文、推理步骤",
|
||||
"brain.late": "后期层 — 规划答案、风格、连贯性",
|
||||
"brain.final": "末尾层 — 输出成型:选择实际的 next-token 分布",
|
||||
|
||||
"profile.title": "性能分析 — 引擎每轮的时间花在哪里",
|
||||
"profile.ioWait": "I/O 等待",
|
||||
"profile.expertMatmul": "专家矩阵乘",
|
||||
"profile.attention": "注意力",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "其他",
|
||||
"profile.empty": "暂无性能数据 — 发送一条消息,分析结果将显示在这里。",
|
||||
"profile.connectHint": "连接引擎以采集每轮耗时。",
|
||||
"profile.lastTurn": "最近一轮",
|
||||
"profile.wallTime": "总耗时",
|
||||
"profile.batching": "批处理",
|
||||
"profile.tokensPerForward": "tokens / 前向",
|
||||
"profile.diskService": "磁盘服务",
|
||||
"profile.overlapped": "与计算重叠",
|
||||
"profile.window": "窗口 · 最近 {{n}} 轮",
|
||||
"profile.throughputTitle": "每轮吞吐量 (tok/s)",
|
||||
"profile.phaseTitle": "每轮各阶段耗时 (s)",
|
||||
"profile.turnCol": "轮次",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "总耗时",
|
||||
"profile.turnsLabel": "{{n}} 轮 · 从旧到新",
|
||||
"profile.oneTurn": "1 轮",
|
||||
"profile.diskNote": "磁盘服务是在 I/O 线程上读取专家的时间;它与计算重叠,因此只有计算线程实际感受到的 I/O 等待 才计入总耗时分解。多 KV 会话时,份额描述的是整个引擎在该轮窗口内的表现。",
|
||||
|
||||
"error.title": "colibrì UI 遇到错误",
|
||||
"error.hint": "引擎不受影响。请尝试刷新页面。",
|
||||
"error.retry": "重试",
|
||||
}
|
||||
|
||||
export default zhCN
|
||||
@@ -0,0 +1,113 @@
|
||||
const zhTW: Record<string, string> = {
|
||||
"nav.chat": "對話",
|
||||
"nav.brain": "大腦",
|
||||
"nav.profiling": "效能分析",
|
||||
|
||||
"brand.tagline": "本地巨人,極小足跡",
|
||||
|
||||
"sidebar.connection": "連線",
|
||||
"sidebar.endpoint": "API 端點",
|
||||
"sidebar.apiKey": "API 金鑰",
|
||||
"sidebar.apiKeyPlaceholder": "選填",
|
||||
"sidebar.apiKeyHelp": "僅保存在記憶體中 · 傳送到此端點",
|
||||
"sidebar.probe": "探測伺服器",
|
||||
"status.connected": "引擎已連線",
|
||||
"status.notConnected": "未連線",
|
||||
"status.runtimeUnavailable": "執行階段指標不可用",
|
||||
"status.serverError": "無法連線到伺服器。",
|
||||
"status.generationFailed": "生成失敗。",
|
||||
|
||||
"sidebar.runtime": "執行階段",
|
||||
"sidebar.runtimeProbe": "探測伺服器以檢視執行階段狀態。",
|
||||
"sidebar.schedulerOnline": "排程器上線",
|
||||
"dashboard.active": "進行中",
|
||||
"dashboard.queued": "排隊中",
|
||||
"dashboard.completed": "已完成",
|
||||
"dashboard.failures": "失敗",
|
||||
"dashboard.session": "工作階段:",
|
||||
"dashboard.prompt": "提示詞",
|
||||
"dashboard.completion": "補全",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "磁碟",
|
||||
"tier.ariaLabel": "專家分佈:{{vram}} VRAM、{{ram}} RAM、{{disk}} 磁碟",
|
||||
|
||||
"sidebar.inference": "推論",
|
||||
"sidebar.model": "模型",
|
||||
"sidebar.kvSession": "KV 工作階段",
|
||||
"sidebar.kvSessionHelp": "獨立上下文 · 對話跟隨所選插槽",
|
||||
"sidebar.sessionLabel": "工作階段 {{slot}}",
|
||||
"sidebar.temperature": "溫度",
|
||||
"sidebar.maxTokens": "最大輸出 token 數",
|
||||
"sidebar.reasoning": "推理模式",
|
||||
"sidebar.transport": "OpenAI 相容協定",
|
||||
|
||||
"topbar.activeModel": "目前模型",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "插槽 {{n}}",
|
||||
"topbar.clear": "清除",
|
||||
|
||||
"hero.title": "COLIBRÌ 引擎",
|
||||
"hero.subtitle": "向巨人提問。",
|
||||
"hero.tagline": "讓機器屬於你。",
|
||||
"hero.description": "連線到本地 colibrì 伺服器,直接從你的硬體串流取得回應。所有資料都留在你選擇的端點內。",
|
||||
"prompts.routing": "解釋專家路由如何運作",
|
||||
"prompts.benchmark": "撰寫一個簡單的 C 基準測試",
|
||||
"prompts.caching": "比較 RAM 與 VRAM 快取",
|
||||
|
||||
"chat.you": "你",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "傳送訊息給 colibrì…",
|
||||
"chat.inputHint": "Enter 傳送 · Shift+Enter 換行",
|
||||
"chat.stop": "停止生成",
|
||||
"chat.send": "傳送訊息",
|
||||
|
||||
"brain.title": "專家皮層",
|
||||
"brain.waiting": "等待引擎連線",
|
||||
"brain.layers": "{{rows}} 層 × {{cols}} 專家",
|
||||
"brain.brightnessHint": "亮度 = 路由熱度",
|
||||
"brain.flashHint": "⚡ 白色閃爍 = 本輪被路由",
|
||||
"brain.connectHint": "連線引擎以檢視皮層。",
|
||||
"brain.neverRouted": "從未被路由",
|
||||
"brain.selections": "約 2^{{heat}} 次選擇",
|
||||
"brain.specialist": "⭐ 專精:{{top}}",
|
||||
"brain.generalist": "通用型",
|
||||
"brain.mtp": "MTP 頭 — 為推測式解碼起草下一個 token",
|
||||
"brain.early": "早期層 — 表面特徵:token、拼寫、局部語法",
|
||||
"brain.lowerMiddle": "中低層 — 片語結構、詞語關係、簡單事實",
|
||||
"brain.upperMiddle": "中高層 — 語意、長距離上下文、推理步驟",
|
||||
"brain.late": "後期層 — 規劃答案、風格、連貫性",
|
||||
"brain.final": "末尾層 — 輸出成型:選擇實際的 next-token 分佈",
|
||||
|
||||
"profile.title": "效能分析 — 引擎每輪的時間花在哪裡",
|
||||
"profile.ioWait": "I/O 等待",
|
||||
"profile.expertMatmul": "專家矩陣乘",
|
||||
"profile.attention": "注意力",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "其他",
|
||||
"profile.empty": "尚無效能數據 — 傳送一則訊息,分析結果將顯示在這裡。",
|
||||
"profile.connectHint": "連線引擎以採集每輪耗時。",
|
||||
"profile.lastTurn": "最近一輪",
|
||||
"profile.wallTime": "總耗時",
|
||||
"profile.batching": "批次處理",
|
||||
"profile.tokensPerForward": "tokens / 前向",
|
||||
"profile.diskService": "磁碟服務",
|
||||
"profile.overlapped": "與運算重疊",
|
||||
"profile.window": "視窗 · 最近 {{n}} 輪",
|
||||
"profile.throughputTitle": "每輪吞吐量 (tok/s)",
|
||||
"profile.phaseTitle": "每輪各階段耗時 (s)",
|
||||
"profile.turnCol": "輪次",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "總耗時",
|
||||
"profile.turnsLabel": "{{n}} 輪 · 從舊到新",
|
||||
"profile.oneTurn": "1 輪",
|
||||
"profile.diskNote": "磁碟服務是在 I/O 執行緒上讀取專家的時間;它與運算重疊,因此只有運算執行緒實際感受到的 I/O 等待 才計入總耗時分解。多 KV 工作階段時,份額描述的是整個引擎在該輪視窗內的表現。",
|
||||
|
||||
"error.title": "colibrì UI 遇到錯誤",
|
||||
"error.hint": "引擎不受影響。請嘗試重新整理頁面。",
|
||||
"error.retry": "重試",
|
||||
}
|
||||
|
||||
export default zhTW
|
||||
+3
-1
@@ -64,7 +64,9 @@ button:focus-visible, input:focus-visible, textarea:focus-visible, select:focus-
|
||||
.toggle-row { display: flex; align-items: center; justify-content: space-between; height: 42px; padding: 0 11px; border: 1px solid var(--border); border-radius: 9px; color: #a9b4b8; background: var(--input); }
|
||||
.toggle-row > span { display: flex; align-items: center; gap: 8px; font-size: 11px; font-weight: 600; }.toggle-row.active { border-color: rgba(78,214,165,.35); color: var(--foreground); }
|
||||
.toggle-row i { width: 30px; height: 17px; padding: 2px; border-radius: 20px; background: #293136; transition: .2s; }.toggle-row i b { display: block; width: 13px; height: 13px; border-radius: 50%; background: #78858a; transition: .2s; }.toggle-row.active i { background: rgba(78,214,165,.28); }.toggle-row.active i b { transform: translateX(13px); background: var(--primary); }
|
||||
.sidebar-foot { margin-top: auto; display: flex; align-items: center; gap: 7px; color: #59666b; font-size: 10px; }
|
||||
.sidebar-foot { margin-top: auto; display: flex; flex-direction: column; gap: 6px; color: #59666b; font-size: 10px; }
|
||||
.sidebar-foot > div { display: flex; align-items: center; gap: 7px; }
|
||||
.locale-switcher select { background: transparent; border: 1px solid var(--border); border-radius: 4px; color: inherit; font-size: 10px; padding: 2px 4px; cursor: pointer; }
|
||||
|
||||
.chat-panel { min-width: 0; height: 100vh; display: grid; grid-template-rows: 72px minmax(0, 1fr) auto; }
|
||||
.topbar { display: flex; align-items: center; justify-content: space-between; padding: 0 32px; border-bottom: 1px solid var(--border); }
|
||||
|
||||
+4
-1
@@ -2,10 +2,13 @@ import { createRoot } from "react-dom/client"
|
||||
|
||||
import App from "./App"
|
||||
import { ErrorBoundary } from "./ErrorBoundary"
|
||||
import { LocaleProvider } from "./i18n"
|
||||
import "./index.css"
|
||||
|
||||
createRoot(document.getElementById("root")!).render(
|
||||
<ErrorBoundary>
|
||||
<App />
|
||||
<LocaleProvider>
|
||||
<App />
|
||||
</LocaleProvider>
|
||||
</ErrorBoundary>,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user