Compare commits
57 Commits
main
...
docs-windows-exe
| Author | SHA1 | Date | |
|---|---|---|---|
| 7de49fa02d | |||
| 505a0f69aa | |||
| 0337697023 | |||
| af23f314fd | |||
| 1765ed80ac | |||
| bb2bae8904 | |||
| 3ddbc655ef | |||
| 3455eeea0a | |||
| 171aa69fcb | |||
| bd3efb1c97 | |||
| f27a89ea82 | |||
| cf126cbcf9 | |||
| d43b54534f | |||
| 42a5417c13 | |||
| ae4e31a15a | |||
| e9b36141a4 | |||
| 9c5ab39b62 | |||
| 4b704823db | |||
| 453d1401ba | |||
| 17fbdfa8f8 | |||
| a7ef058fc0 | |||
| 5c84b25645 | |||
| fbabaa4544 | |||
| 1f00142b25 | |||
| 26bd8b403a | |||
| 70fb5b00f3 | |||
| 4586d33c60 | |||
| 61004dcb84 | |||
| 8a9a0fca4d | |||
| 083fda5b0a | |||
| bc69a9a6d0 | |||
| 93b4a8e78e | |||
| e486574442 | |||
| 420a0720c3 | |||
| f853ea8a0b | |||
| 8f33bd153b | |||
| cfcc742591 | |||
| ebc851edb3 | |||
| 845af6378d | |||
| 3ffe4bb75e | |||
| 72e36772f5 | |||
| fae4b2cc3e | |||
| 22509fccde | |||
| ca39e5333f | |||
| 741d46ba25 | |||
| d86a6b93ad | |||
| c769e04d13 | |||
| b6bae91b66 | |||
| 2d8d2951ee | |||
| 1ac2e7b487 | |||
| 6ade4093de | |||
| ca788833ab | |||
| 24058d3de8 | |||
| b3fdb145f8 | |||
| 4ae4f61d16 | |||
| c4624276f3 | |||
| ac1f7a8f38 |
@@ -0,0 +1,10 @@
|
||||
BasedOnStyle: LLVM
|
||||
IndentWidth: 4
|
||||
ColumnLimit: 120
|
||||
AllowShortFunctionsOnASingleLine: All
|
||||
AllowShortIfStatementsOnASingleLine: AllIfsAndElse
|
||||
AllowShortLoopsOnASingleLine: true
|
||||
BreakBeforeBraces: Attach
|
||||
PointerAlignment: Right
|
||||
SpaceAfterCStyleCast: false
|
||||
SortIncludes: false
|
||||
@@ -0,0 +1,24 @@
|
||||
root = true
|
||||
|
||||
[*]
|
||||
charset = utf-8
|
||||
end_of_line = lf
|
||||
insert_final_newline = true
|
||||
trim_trailing_whitespace = true
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
|
||||
[*.{c,h,cu}]
|
||||
indent_size = 4
|
||||
|
||||
[*.{ts,tsx,js,json,css}]
|
||||
indent_size = 2
|
||||
|
||||
[Makefile]
|
||||
indent_style = tab
|
||||
|
||||
[*.yml]
|
||||
indent_size = 2
|
||||
|
||||
[*.md]
|
||||
trim_trailing_whitespace = false
|
||||
@@ -12,8 +12,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Build glm
|
||||
run: cd c && make glm
|
||||
- name: Build colibri
|
||||
run: cd c && make colibri
|
||||
- name: C test suite
|
||||
run: cd c && make test-c
|
||||
|
||||
@@ -105,13 +105,13 @@ jobs:
|
||||
make cuda-dll CUDA_ARCH=sm_80
|
||||
test -f coli_cuda.dll || { echo "cuda-dll reported success but produced no DLL" >&2; exit 1; }
|
||||
echo "coli_cuda.dll built (MSVC host)"
|
||||
- name: make glm CUDA_DLL=1 (host links backend_loader, not cudart)
|
||||
- name: make colibri CUDA_DLL=1 (host links backend_loader, not cudart)
|
||||
shell: msys2 {0}
|
||||
run: |
|
||||
cd c
|
||||
make glm CUDA_DLL=1
|
||||
test -f glm.exe || { echo "glm CUDA_DLL=1 reported success but produced no exe" >&2; exit 1; }
|
||||
echo "glm.exe built against the DLL loader"
|
||||
make colibri CUDA_DLL=1
|
||||
test -f colibri.exe || { echo "colibri CUDA_DLL=1 reported success but produced no exe" >&2; exit 1; }
|
||||
echo "colibri.exe built against the DLL loader"
|
||||
|
||||
web:
|
||||
name: Web UI
|
||||
|
||||
@@ -10,6 +10,8 @@ desktop/src-tauri/target/
|
||||
desktop/src-tauri/gen/
|
||||
|
||||
# binari compilati (si rigenerano con make / coli build)
|
||||
c/colibri
|
||||
c/colibri.exe
|
||||
c/glm
|
||||
c/glm.exe
|
||||
c/olmoe
|
||||
@@ -68,3 +70,7 @@ c/tests/test_decode_batch
|
||||
c/tests/test_i4_acc512
|
||||
c/tests/test_idot
|
||||
c/tests/test_uring
|
||||
olmoe_merged/
|
||||
olmoe_i4/
|
||||
c/olmoe_merged/
|
||||
c/olmoe_i4/
|
||||
|
||||
+260
@@ -0,0 +1,260 @@
|
||||
<p align="center">
|
||||
<img src="assets/colibri.svg" width="500" alt="colibrì — motore piccolo, modello immenso">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a> · <a href="README.zh-TW.md">繁體中文</a> · Italiano
|
||||
</p>
|
||||
|
||||
**Motore piccolo, modello immenso.** Esegui **GLM-5.2 (744 miliardi di parametri, MoE)** su un computer consumer con ~25 GB di RAM — in C puro, zero dipendenze, caricando gli expert dal disco in streaming.
|
||||
|
||||
Colibrì è un runtime MoE leggero e che preserva la qualità: tratta VRAM, RAM e
|
||||
disco come un'unica gerarchia di memoria gestita. Se la memoria veloce non basta
|
||||
il modello rallenta, ma la policy predefinita **non cambia mai silenziosamente la
|
||||
precisione del modello né la semantica del router**.
|
||||
|
||||
```
|
||||
$ ./coli chat
|
||||
🐦 colibrì v1.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
|
||||
✓ ready in 32s · resident 9.9 GB
|
||||
› ciao!
|
||||
◆ Ciao! 😊 Come posso aiutarti oggi?
|
||||
```
|
||||
|
||||
## Guardalo in azione
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-dashboard.png" width="900" alt="dashboard web di colibrì — metriche live, pannello hardware, livelli degli expert">
|
||||
</p>
|
||||
<p align="center"><em>La dashboard web (<code>./coli web</code>): un modello da 744B a <strong>4 tok/s, TTFT 1.6 s, disco 0</strong> —
|
||||
residenza completa degli expert su 6× RTX 5090, con metriche token in tempo reale, breakdown dei tempi per turno,
|
||||
la barra dei livelli VRAM/RAM/disco e il mini-cervello live nell'angolo.</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-brain.png" width="900" alt="la pagina Brain — 19.456 expert come una corteccia vivente">
|
||||
</p>
|
||||
<p align="center"><em>La pagina <strong>Brain</strong>: tutti i 19.456 expert come una corteccia vivente — il colore indica
|
||||
il livello di archiviazione, la luminosità il calore di routing, e ogni expert instradato in un turno
|
||||
lampeggia bianco. Passando il cursore si vede l'<a href="https://github.com/JustVugg/colibri/issues/175">affinità
|
||||
tematica misurata</a> dell'expert.</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-atlas.png" width="900" alt="la pagina Atlas — l'atlante misurato degli expert come una galassia 3D">
|
||||
</p>
|
||||
<p align="center"><em>La pagina <strong>Atlas</strong>: l'<a href="https://github.com/JustVugg/colibri/issues/175">atlante
|
||||
misurato degli expert</a> come una galassia 3D — 13.260 expert caratterizzati, 1.041 specialisti
|
||||
replicabili che si raggruppano per argomento (poesia, legge, cinese, SQL…). La posizione deriva
|
||||
dall'affinità di routing misurata, non da un embedding appreso. Trascinare per ruotare.</em></p>
|
||||
|
||||
## L'idea
|
||||
|
||||
Un modello Mixture-of-Experts da 744B attiva solo ~40B parametri per token — e
|
||||
solo ~11 GB di quelli cambiano da un token all'altro (gli expert instradati):
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/sparse.png" width="880" alt="solo ~5.4% dei parametri è attivo per token">
|
||||
</p>
|
||||
|
||||
Il modello non ha bisogno di *stare* in memoria veloce — ha bisogno di essere
|
||||
**piazzato**:
|
||||
|
||||
- la **parte densa** (attenzione, expert condivisi, embedding — ~17B parametri)
|
||||
resta **residente in RAM a int4** (~9.9 GB);
|
||||
- i **19.456 expert instradati** (75 layer MoE × 256 + la testa MTP, ~19 MB
|
||||
ciascuno a int4) stanno **su disco** (~370 GB) e vengono **caricati on demand
|
||||
in streaming**, con una cache LRU per layer, un hot-store pinnato che impara,
|
||||
e un livello VRAM opzionale.
|
||||
|
||||
Il motore è un singolo file C (`c/colibri.c`) più header piccoli. Niente BLAS,
|
||||
niente Python a runtime, niente GPU obbligatoria.
|
||||
|
||||
## Come funziona
|
||||
|
||||
### Il percorso di ogni token
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/token-path.png" width="880" alt="instrada → unione → piazza → sovrapponi → impara">
|
||||
</p>
|
||||
|
||||
Ogni layer di ogni token percorre gli stessi cinque passi. L'obiettivo
|
||||
progettuale è che **il piazzamento decide solo la velocità** — le decisioni
|
||||
del router e la precisione dei pesi sono identiche sia che un expert risponda
|
||||
dalla VRAM sia dal disco.
|
||||
|
||||
### Una gerarchia di memoria, non un requisito di memoria
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/tiers.png" width="880" alt="residenza expert a tre livelli: VRAM / RAM / NVMe">
|
||||
</p>
|
||||
|
||||
Lo stesso motore copre l'intero spettro: su un portatile da 25 GB tutto viene
|
||||
caricato dal disco in streaming (lento, ma corretto); su un host grande l'intero
|
||||
set di expert diventa residente (`CUDA_EXPERT_GB=auto PIN_GB=all`) e il disco
|
||||
esce completamente dal percorso di decode. Tra i livelli c'è una **cache che
|
||||
impara**: il motore registra quali expert il *tuo* carico di lavoro instrada
|
||||
(`.coli_usage`, aggiornato a ogni turno) e fissa automaticamente i più caldi —
|
||||
colibrì diventa letteralmente più veloce man mano che lo usi. Sugli host
|
||||
multi-socket, `COLI_NUMA=1` interlaccia i pesi residenti tra i controller di
|
||||
memoria ([#82](https://github.com/JustVugg/colibri/issues/82)).
|
||||
|
||||
### Mai aspettare il disco due volte
|
||||
|
||||
I miss nella cache costano caro, quindi il motore investe la maggior parte
|
||||
della sua astuzia per evitarli e sovrapporli: le tre matrici di ogni expert sono
|
||||
memorizzate contigue e lette con un unico `pread`; un pool I/O asincrono
|
||||
limitato (`PIPE=1`, attivo per default) carica gli expert mancanti mentre quelli
|
||||
residenti calcolano; le posizioni in batch leggono ogni expert unico una sola
|
||||
volta (**batch-union**); un thread di lookahead del router (`PILOT=1`) fa il
|
||||
prefetch degli expert del layer successivo — il routing è misurabilmente
|
||||
**prevedibile al 71.6% un layer in anticipo**. Sulle GPU, la pipeline residente
|
||||
(`COLI_CUDA_PIPE=2`) mantiene il flusso residuo on-device tra i layer, così il
|
||||
loop CPU degli expert procede senza interruzioni; su Apple Silicon un backend
|
||||
[Metal](docs/metal.md) sperimentale esegue la matmul batch degli expert sulla
|
||||
GPU a memoria unificata.
|
||||
|
||||
### Modello fedele, stato compresso
|
||||
|
||||
Il forward pass è validato **token-esatto contro un oracle `transformers`**
|
||||
(teacher-forcing 32/32). L'attenzione MLA memorizza uno stato KV compresso — 576
|
||||
float/token invece di 32.768 (**57× più piccolo**) — e lo persiste tra i
|
||||
riavvii (`.coli_kv`): le conversazioni riaprono "calde", senza alcun re-prefill,
|
||||
byte-identiche a una sessione ininterrotta. L'attenzione sparsa DSA (il
|
||||
lightning indexer di GLM-5.2) è implementata fedelmente e validata forzando la
|
||||
selezione di tutte le chiavi per riprodurre esattamente l'attenzione densa.
|
||||
|
||||
### Decodifica speculativa, onestamente
|
||||
|
||||
La testa MTP nativa di GLM-5.2 propone token che il modello principale verifica
|
||||
in un unico forward batch — 2.2–2.8 token/forward quando conviene. Due regole
|
||||
conquistate a caro prezzo sono i default: la testa MTP deve essere **int8** (le
|
||||
teste int4 crollano al 0–4% di accettazione,
|
||||
[#8](https://github.com/JustVugg/colibri/issues/8)), e draft e verifica devono
|
||||
calcolare **la stessa funzione** — `SPEC_PIN=1` fissa entrambi sulla stessa
|
||||
famiglia di kernel ([#163](https://github.com/JustVugg/colibri/issues/163)
|
||||
contiene l'intera indagine forense). I draft forzati da grammatica
|
||||
([`GRAMMAR=file.gbnf`](docs/grammar-draft.md)) aggiungono accettazione quasi
|
||||
gratuita sull'output JSON vincolato. Se la speculazione conviene dipende dalla
|
||||
temperatura della cache — misura, e usa `DRAFT=0` quando non paga.
|
||||
|
||||
## Cosa ottiene
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/ladder.png" width="880" alt="velocità di decode misurata per classe hardware">
|
||||
</p>
|
||||
|
||||
Stesso motore, stesso container int4 — cambia solo dove risiedono gli expert.
|
||||
Punti salienti dalle [tabelle benchmark complete](docs/benchmarks.md):
|
||||
|
||||
- **6× RTX 5090, residenza completa:** 5.8–6.8 tok/s in decode, TTFT ~13 s
|
||||
([log dell'esperimento](docs/experiments/glm52-6x5090-2026-07-12.md));
|
||||
- **desktop solo-CPU da 128 GB:** ~1.8 tok/s a cache calda
|
||||
([#200](https://github.com/JustVugg/colibri/issues/200));
|
||||
- **singola RTX 5070 Ti, classe laptop:** 1.07 tok/s tramite la pipeline
|
||||
GPU-residente ([#273](https://github.com/JustVugg/colibri/issues/273));
|
||||
- **macchina di sviluppo da 25 GB:** 0.05–0.1 tok/s a freddo — il punto di
|
||||
partenza dimostrato da cui è nato il progetto, e ancora oggi la baseline onesta.
|
||||
|
||||
La qualità è misurata, non presunta: il costo di quantizzazione del container
|
||||
int4 e le ablazioni su granularità delle scale e rotazione sono in
|
||||
[docs/benchmarks.md](docs/benchmarks.md#quality-benchmark) e
|
||||
[#108](https://github.com/JustVugg/colibri/issues/108)/[#81](https://github.com/JustVugg/colibri/issues/81).
|
||||
|
||||
## Per iniziare
|
||||
|
||||
### 1. Scarica il modello
|
||||
|
||||
Un container **GLM-5.2 int4** pre-convertito è su Hugging Face — **usa la
|
||||
versione con le teste MTP int8**:
|
||||
|
||||
**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp**
|
||||
|
||||
> ⚠️ Il mirror originale contiene teste MTP int4 → accettazione dei draft allo 0%
|
||||
> ([#8](https://github.com/JustVugg/colibri/issues/8)). Verifica la tua versione:
|
||||
> `ls -l <modello>/out-mtp-*` — int8 (corretto) è `3527131672 / 5366238584 / 1065950496`.
|
||||
|
||||
Oppure converti tu stesso dalla sorgente FP8 — un unico comando riprendibile che
|
||||
non richiede mai i 756 GB completi su disco contemporaneamente:
|
||||
|
||||
```bash
|
||||
cd c && ./setup.sh # verifica gcc/OpenMP, compila, autotest
|
||||
./coli convert --model /nvme/glm52_i4 # scarica e converti shard per shard (python, una tantum)
|
||||
```
|
||||
|
||||
### 2. Esegui
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat # budget RAM, cache e MTP rilevati automaticamente
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli plan # mostra il piazzamento pianificato VRAM/RAM/disco
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli doctor # controllo di idoneità (sola lettura)
|
||||
./coli web --model /nvme/glm52_i4 # API + dashboard web sulla stessa porta
|
||||
./coli serve --model /nvme/glm52_i4 # solo API compatibile OpenAI
|
||||
```
|
||||
|
||||
Il motore a runtime è puro C — python si usa solo per il convertitore (una tantum)
|
||||
e per il gateway API opzionale.
|
||||
|
||||
### 3. Approfondisci
|
||||
|
||||
| argomento | documento |
|
||||
|---|---|
|
||||
| Benchmark, dati dalla comunità, misurazioni di qualità | [docs/benchmarks.md](docs/benchmarks.md) |
|
||||
| Parametri di tuning, policy, cache che impara, prefetch | [docs/tuning.md](docs/tuning.md) |
|
||||
| Build nativa su Windows 11 (con CUDA DLL) | [docs/windows.md](docs/windows.md) |
|
||||
| Backend CUDA, livello expert in VRAM, residenza completa | [docs/cuda.md](docs/cuda.md) |
|
||||
| Backend Metal per Apple Silicon | [docs/metal.md](docs/metal.md) |
|
||||
| API compatibile OpenAI, KV slot, dashboard web | [docs/api.md](docs/api.md) |
|
||||
| Draft forzati da grammatica (output strutturato) | [docs/grammar-draft.md](docs/grammar-draft.md) |
|
||||
| Inventario delle variabili d'ambiente | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
|
||||
|
||||
## Sostenere il progetto
|
||||
|
||||
colibrì è nato come progetto di una sola persona su un portatile con 12 core
|
||||
e 25 GB di RAM; oggi i suoi numeri arrivano da una comunità di macchine reali.
|
||||
Se ti è utile:
|
||||
|
||||
- ⭐ metti una stella al repository e condividilo;
|
||||
- 🐛 apri issue con i numeri di benchmark del tuo hardware — i datapoint
|
||||
fanno avanzare questo progetto più di qualsiasi altra cosa;
|
||||
- 💬 contattaci via GitHub issues per sponsorizzare lo sviluppo o donare hardware.
|
||||
|
||||
## Struttura del repository
|
||||
|
||||
```
|
||||
Makefile punto d'ingresso root per build/check
|
||||
c/
|
||||
├── colibri.c motore principale
|
||||
├── quant.h kernel matmul quantizzati (SIMD multi-architettura)
|
||||
├── sample.h campionamento, RNG, set di stop
|
||||
├── kv_persist.h persistenza KV su disco (.coli_kv)
|
||||
├── telemetry.h protocollo dashboard, statistiche, usage
|
||||
├── st.h, tok.h, json.h header di runtime
|
||||
├── backend_cuda.* livello CUDA opzionale
|
||||
├── Makefile build e check locali
|
||||
├── coli CLI utente
|
||||
├── openai_server.py gateway HTTP compatibile OpenAI
|
||||
├── setup.sh setup locale in un solo comando
|
||||
├── tools/ conversione offline, fixture e benchmark
|
||||
├── scripts/ helper per conversioni lunghe
|
||||
└── tests/ test C e Python senza dipendenze
|
||||
web/ UI browser (puro client API OpenAI)
|
||||
desktop/ shell desktop Tauri v2 che racchiude la web UI
|
||||
docs/ documentazione di riferimento, esperimenti, media
|
||||
```
|
||||
|
||||
Il percorso a runtime resta intenzionalmente piatto e leggibile: `colibri.c`
|
||||
più i suoi header. Dalla radice del repository, `make`, `make check` e
|
||||
`make clean` delegano al Makefile del motore.
|
||||
|
||||
## Perché "colibrì"
|
||||
|
||||
Il colibrì pesa pochi grammi, sta sospeso nel vuoto e visita un migliaio di
|
||||
fiori al giorno. Questo motore tiene in vita un gigante da 744 miliardi di
|
||||
parametri con le razioni di un colibrì: 25 GB di RAM, dodici core CPU e
|
||||
tanta pazienza col disco.
|
||||
|
||||
Il nome è rimasto in italiano perché questa è la lingua in cui è stato scritto
|
||||
il primo prototipo — i commenti nel codice lo testimoniano ancora.
|
||||
|
||||
## Licenza
|
||||
|
||||
Apache 2.0. I pesi di GLM-5.2 sono rilasciati da Z.ai sotto licenza MIT.
|
||||
@@ -3,7 +3,7 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
English · <a href="README.zh-TW.md">繁體中文</a>
|
||||
English · <a href="README.zh-CN.md">简体中文</a> · <a href="README.zh-TW.md">繁體中文</a> · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**Tiny engine, immense model.** Run **GLM-5.2 (744B-parameter MoE)** on a consumer machine with ~25 GB of RAM — in pure C, with zero dependencies, by streaming experts from disk.
|
||||
@@ -151,6 +151,10 @@ scale-granularity/rotation ablations live in
|
||||
|
||||
## Get started
|
||||
|
||||
> **New here?** The [Quick Start guide](docs/quickstart.md) walks through
|
||||
> install → build → model → first chat step by step for Linux, Windows, and
|
||||
> macOS, with copy-paste commands and no assumed background.
|
||||
|
||||
### 1. Get the model
|
||||
|
||||
A pre-converted **GLM-5.2 int4** container is on Hugging Face — **use the
|
||||
@@ -183,6 +187,17 @@ COLI_MODEL=/nvme/glm52_i4 ./coli doctor # read-only readiness check
|
||||
The engine at runtime is pure C — python is only used by the one-time converter
|
||||
and the optional API gateway.
|
||||
|
||||
**On Windows?** You don't need to build. Download the
|
||||
`colibri-<version>-windows-x86_64.zip` from
|
||||
[Releases](https://github.com/JustVugg/colibri/releases), unzip it, rename
|
||||
`colibri-*-windows-x86_64.exe` → `glm.exe` (so the `coli` launcher finds the
|
||||
engine), install [Python 3](https://www.python.org/downloads/), then run
|
||||
`coli chat`. Full walkthrough in the [Quick Start guide](docs/quickstart.md#windows).
|
||||
|
||||
Prefer a `coli` command on your PATH? From a checkout, `pip install -e .`
|
||||
registers it (the engine itself still lives in `c/` — this is an editable
|
||||
install from the clone, not a standalone wheel).
|
||||
|
||||
### 3. Go deeper
|
||||
|
||||
| topic | doc |
|
||||
|
||||
+237
@@ -0,0 +1,237 @@
|
||||
<p align="center">
|
||||
<img src="assets/colibri.svg" width="500" alt="colibrì——小巧引擎,庞大模型">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · 简体中文 · <a href="README.zh-TW.md">繁體中文</a> · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**小巧引擎,庞大模型。**只需约 25 GB 内存,就能在消费级电脑上运行 **GLM-5.2(744B 参数的 MoE)**——以零依赖的纯 C 实现,从磁盘流式加载专家。
|
||||
|
||||
Colibrì 是一套轻量、保持模型质量的 MoE 运行时,将 VRAM、RAM
|
||||
与存储设备视为统一管理的内存层级。高速内存不足可能降低速度,
|
||||
但默认策略**绝不会在未告知的情况下改变模型精度或路由语义**。
|
||||
|
||||
```
|
||||
$ ./coli chat
|
||||
🐦 colibrì v1.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
|
||||
✓ ready in 32s · resident 9.9 GB
|
||||
› ciao!
|
||||
◆ Ciao! 😊 Come posso aiutarti oggi?
|
||||
```
|
||||
|
||||
## 实际运行效果
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-dashboard.png" width="900" alt="colibrì 网页仪表盘——实时指标、硬件面板与专家存储层级">
|
||||
</p>
|
||||
<p align="center"><em>网页仪表盘(<code>./coli web</code>):744B 模型达到 <strong>4 tok/s、TTFT 1.6 秒、磁盘读取 0</strong>——
|
||||
在 6× RTX 5090 上让所有专家常驻,并实时显示 token 指标、每轮耗时明细、
|
||||
VRAM/RAM/磁盘层级条,以及角落的实时迷你大脑。</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-brain.png" width="900" alt="大脑页面——以实时皮层呈现 19,456 个专家">
|
||||
</p>
|
||||
<p align="center"><em><strong>大脑(Brain)</strong>页面:将全部 19,456 个专家呈现为活的皮层——颜色代表存储层级,
|
||||
亮度代表路由热度,每轮被路由到的专家都会闪白。将光标停在专家上,即可查看其
|
||||
<a href="https://github.com/JustVugg/colibri/issues/175">实测主题亲和度</a>。</em></p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/colibri-atlas.png" width="900" alt="图谱页面——以 3D 星系呈现实测专家图谱">
|
||||
</p>
|
||||
<p align="center"><em><strong>图谱(Atlas)</strong>页面:将<a href="https://github.com/JustVugg/colibri/issues/175">实测专家图谱</a>
|
||||
呈现为 3D 星系——共 13,260 个已分析专家,其中 1,041 个可复现的专门专家会按主题聚集
|
||||
(诗歌、法律、中文、SQL……)。位置取自实测路由亲和度,而非学习出的嵌入向量。拖拽即可旋转。</em></p>
|
||||
|
||||
## 核心概念
|
||||
|
||||
744B 的专家混合(Mixture-of-Experts)模型,每个 token 只会激活约 40B 参数——
|
||||
其中每个 token 之间会变动的只有约 11 GB(被路由到的专家):
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/sparse.png" width="880" alt="每个 token 只会激活约 5.4% 的参数">
|
||||
</p>
|
||||
|
||||
所以模型不必完整**装进**高速内存,而是需要正确**放置**:
|
||||
|
||||
- **稠密部分**(注意力、共享专家、嵌入——约 17B 参数)以 int4
|
||||
**常驻 RAM**(约 9.9 GB);
|
||||
- **19,456 个路由专家**(75 个 MoE 层 × 256,加上 MTP head;每个在 int4 下约 19 MB)
|
||||
**存放在磁盘**(约 370 GB),并**按需流式加载**,配合逐层 LRU 缓存、
|
||||
会学习的热门专家固定存储区,以及可选的 VRAM 层级。
|
||||
|
||||
引擎是一个 C 主文件(`c/colibri.c`)加上若干头文件。不需要 BLAS,
|
||||
运行时不需要 Python,也不需要 GPU。
|
||||
|
||||
## 工作原理
|
||||
|
||||
### 每个 token 的处理路径
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/token-path.png" width="880" alt="路由 → 并集 → 放置 → 重叠执行 → 学习">
|
||||
</p>
|
||||
|
||||
每个 token 的每一层都会经过相同的五个步骤。设计目标是让
|
||||
**放置只决定速度**——无论专家是从 VRAM 还是磁盘响应,路由器的决策与权重精度都完全相同。
|
||||
|
||||
### 统一内存层级,取代单一内存门槛
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/tiers.png" width="880" alt="VRAM/RAM/NVMe 三层专家常驻架构">
|
||||
</p>
|
||||
|
||||
同一套引擎覆盖完整硬件范围:在 25 GB 笔记本上,一切都从磁盘流式加载
|
||||
(慢,但结果正确);在大内存主机上,则可让整组专家常驻
|
||||
(`CUDA_EXPERT_GB=auto PIN_GB=all`),让磁盘完全退出解码路径。
|
||||
两端之间有一层**学习型缓存**:引擎会记录*你的*工作负载路由到哪些专家
|
||||
(`.coli_usage`,每轮更新),并自动固定最热门的专家——colibrì 确实会越用越快。
|
||||
在多路主机上,`COLI_NUMA=1` 会将常驻权重交错分配到各内存控制器
|
||||
([#82](https://github.com/JustVugg/colibri/issues/82))。
|
||||
|
||||
### 绝不为同一次磁盘读取等待两遍
|
||||
|
||||
缓存未命中的代价很高,因此引擎大部分的巧思都用来避免或重叠这些读取:
|
||||
每个专家的三个矩阵相邻存储,并以一次 `pread` 读取;有界异步 I/O 池
|
||||
(`PIPE=1`,默认启用)会在常驻专家计算时加载缺失的专家;批量位置只读取每个
|
||||
不重复专家一次(**批量并集**);路由前瞻线程(`PILOT=1`)则预取下一层专家——
|
||||
实测显示,路由结果提前一层时有 **71.6% 的可预测性**。
|
||||
在 GPU 上,常驻管线(`COLI_CUDA_PIPE=2`)让残差流跨层保留在设备端,
|
||||
使 CPU 专家循环不中断;在 Apple Silicon 上,实验性的
|
||||
[Metal 后端](docs/metal.md)会用统一内存 GPU 执行批量专家运算。
|
||||
|
||||
### 忠实模型,压缩状态
|
||||
|
||||
前向传播已通过 `transformers` oracle 验证为**逐 token 完全一致**
|
||||
(teacher-forcing 32/32)。MLA 注意力存储压缩后的 KV 状态——每个 token 为 576 个
|
||||
浮点数,而非 32,768 个(**缩小 57×**)——并跨重启持久保存
|
||||
(`.coli_kv`):对话可暖启恢复,不需重新 prefill,结果与不中断的会话
|
||||
逐字节相同。DSA 稀疏注意力(GLM-5.2 的 lightning indexer)已忠实实现,
|
||||
并通过强制选取所有 key,验证可精确复现稠密注意力。
|
||||
|
||||
### 诚实的推测解码
|
||||
|
||||
GLM-5.2 原生 MTP head 会起草 token,再由主模型以一次批量前向传播验证——
|
||||
条件合适时每次 forward 可产生 2.2–2.8 个 token。两条来之不易的规则已成为默认值:
|
||||
MTP head 必须是 **int8**(int4 head 的接受率会崩塌到 0–4%,见
|
||||
[#8](https://github.com/JustVugg/colibri/issues/8)),且草稿与验证必须计算
|
||||
**相同函数**——`SPEC_PIN=1` 会把两者固定在同一 kernel family
|
||||
(完整取证过程见 [#163](https://github.com/JustVugg/colibri/issues/163))。
|
||||
语法强制草稿([`GRAMMAR=file.gbnf`](docs/grammar-draft.md))可在受限 JSON 输出中,
|
||||
以近乎免费的代价提高接受率。推测解码是否带来净收益取决于缓存热度——请实测,
|
||||
若不划算就使用 `DRAFT=0`。
|
||||
|
||||
## 实际成果
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/media/ladder.png" width="880" alt="各硬件级别的实测解码速度">
|
||||
</p>
|
||||
|
||||
同一套引擎、同一个 int4 容器——硬件只会改变专家的存放位置。
|
||||
[完整 benchmark 表格](docs/benchmarks.md)中的重点如下:
|
||||
|
||||
- **6× RTX 5090,全部常驻:**解码 5.8–6.8 tok/s,TTFT 约 13 秒
|
||||
([实验记录](docs/experiments/glm52-6x5090-2026-07-12.md));
|
||||
- **128 GB、仅使用 CPU 的台式机:**热缓存后约 1.8 tok/s
|
||||
([#200](https://github.com/JustVugg/colibri/issues/200));
|
||||
- **单张 RTX 5070 Ti 的笔记本级主机:**通过 GPU 常驻管线达到 1.07 tok/s
|
||||
([#273](https://github.com/JustVugg/colibri/issues/273));
|
||||
- **25 GB 开发机:**冷启动 0.05–0.1 tok/s——这是项目起步时已证实的下限,
|
||||
也仍是诚实的基准。
|
||||
|
||||
质量来自测量,而非假设:int4 容器的量化损失,以及 scale granularity/rotation
|
||||
消融实验,收录于 [docs/benchmarks.md](docs/benchmarks.md#quality-benchmark)、
|
||||
[#108](https://github.com/JustVugg/colibri/issues/108) 与
|
||||
[#81](https://github.com/JustVugg/colibri/issues/81)。
|
||||
|
||||
## 开始使用
|
||||
|
||||
### 1. 获取模型
|
||||
|
||||
Hugging Face 上已有预转换的 **GLM-5.2 int4** 容器——请务必使用
|
||||
**含 int8 MTP head 的版本**:
|
||||
|
||||
**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp**
|
||||
|
||||
> ⚠️ 原始镜像使用 int4 MTP head → 草稿接受率为 0%
|
||||
>([#8](https://github.com/JustVugg/colibri/issues/8))。请检查你的版本:
|
||||
> `ls -l <model>/out-mtp-*`——正确的 int8 大小为 `3527131672 / 5366238584 / 1065950496`。
|
||||
|
||||
你也可以自行从 FP8 源转换——只需一条可断点续传的命令,且任何时候都不需要
|
||||
在磁盘上同时存放完整的 756 GB:
|
||||
|
||||
```bash
|
||||
cd c && ./setup.sh # 检查 gcc/OpenMP、构建并运行自测
|
||||
./coli convert --model /nvme/glm52_i4 # 逐 shard 下载并转换(仅此一次需要 python)
|
||||
```
|
||||
|
||||
### 2. 运行
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat # 自动检测 RAM 预算、缓存与 MTP
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli plan # 查看规划的 VRAM/RAM/磁盘配置
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli doctor # 只读就绪检查
|
||||
./coli web --model /nvme/glm52_i4 # 在同一端口提供 API 与网页仪表盘
|
||||
./coli serve --model /nvme/glm52_i4 # 仅提供 OpenAI 兼容 API
|
||||
```
|
||||
|
||||
引擎运行时是纯 C——python 只供一次性转换工具与可选的 API gateway 使用。
|
||||
|
||||
### 3. 深入了解
|
||||
|
||||
| 主题 | 文档 |
|
||||
|---|---|
|
||||
| Benchmark、社区实测数据、质量测量 | [docs/benchmarks.md](docs/benchmarks.md) |
|
||||
| 调优选项、策略、学习型缓存、预取 | [docs/tuning.md](docs/tuning.md) |
|
||||
| Windows 11 原生构建(含 CUDA DLL) | [docs/windows.md](docs/windows.md) |
|
||||
| CUDA 后端、VRAM 专家层级、全部常驻 | [docs/cuda.md](docs/cuda.md) |
|
||||
| Apple Silicon Metal 后端 | [docs/metal.md](docs/metal.md) |
|
||||
| OpenAI 兼容 API、KV slots、网页仪表盘 | [docs/api.md](docs/api.md) |
|
||||
| 语法强制草稿(结构化输出) | [docs/grammar-draft.md](docs/grammar-draft.md) |
|
||||
| 环境变量完整清单 | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
|
||||
|
||||
## 支持项目
|
||||
|
||||
colibrì 最初由一人使用 12 核心、25 GB RAM 的笔记本开发;
|
||||
如今它的数据来自社区中各种真实机器。如果这个项目对你有用:
|
||||
|
||||
- ⭐ 为仓库加星并分享;
|
||||
- 🐛 以 issue 提交你的硬件 benchmark 数据——实测数据比任何其他事都更能推动项目;
|
||||
- 💬 若想赞助开发或捐赠硬件,请通过 GitHub issues 联系。
|
||||
|
||||
## 仓库结构
|
||||
|
||||
```
|
||||
Makefile 根目录构建/检查入口
|
||||
c/
|
||||
├── colibri.c 引擎主文件
|
||||
├── quant.h 量化 matmul 内核(SIMD 多架构)
|
||||
├── sample.h 采样与 stop-set 管理
|
||||
├── kv_persist.h .coli_kv 磁盘持久化
|
||||
├── telemetry.h 仪表盘协议、统计与用量持久化
|
||||
├── st.h, tok.h, json.h 运行时头文件
|
||||
├── backend_cuda.* 可选的 CUDA 层级
|
||||
├── Makefile 构建与本地检查
|
||||
├── coli 用户界面 CLI
|
||||
├── openai_server.py OpenAI 兼容 HTTP gateway
|
||||
├── setup.sh 一条命令完成本地设置
|
||||
├── tools/ 离线转换、fixtures 与 benchmarks
|
||||
├── scripts/ 长时间转换辅助工具
|
||||
└── tests/ 零依赖的 C 与 Python 测试
|
||||
web/ 浏览器 UI(纯 OpenAI API client)
|
||||
desktop/ 封装网页 UI 的 Tauri v2 桌面 shell
|
||||
docs/ 参考文档、实验与媒体文件
|
||||
```
|
||||
|
||||
运行时路径刻意保持扁平、易读:`colibri.c` 加上若干头文件。
|
||||
在仓库根目录执行 `make`、`make check` 与 `make clean`,
|
||||
都会转发给引擎的 Makefile。
|
||||
|
||||
## 为什么叫"colibrì"
|
||||
|
||||
蜂鸟只有几克重,能在原地悬停,并在一天内造访上千朵花。
|
||||
这套引擎只用蜂鸟般的配给,就能让 744B 参数的巨人运转:
|
||||
25 GB RAM、十二个 CPU 核心,以及对磁盘的大量耐心。
|
||||
|
||||
## 许可证
|
||||
|
||||
Apache 2.0。GLM-5.2 权重由 Z.ai 以 MIT 许可发布。
|
||||
+8
-4
@@ -3,7 +3,7 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="README.md">English</a> · 繁體中文
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a> · 繁體中文 · <a href="README.it.md">Italiano</a>
|
||||
</p>
|
||||
|
||||
**小巧引擎,龐大模型。**只要約 25 GB 記憶體,就能在消費級電腦上執行 **GLM-5.2(744B 參數的 MoE)**——以零相依套件的純 C 實作,從硬碟串流載入專家。
|
||||
@@ -60,7 +60,7 @@ VRAM/RAM/硬碟層級長條,以及角落的即時迷你大腦。</em></p>
|
||||
**存放在硬碟**(約 370 GB),並**隨需串流載入**,搭配逐層 LRU 快取、
|
||||
會學習的熱門專家固定儲存區,以及選用的 VRAM 層級。
|
||||
|
||||
引擎由單一 C 檔(`c/glm.c`)與少量標頭檔組成。不需要 BLAS,
|
||||
引擎由主 C 檔(`c/colibri.c`)與多個標頭檔模組組成。不需要 BLAS,
|
||||
執行階段不需要 Python,也不需要 GPU。
|
||||
|
||||
## 運作方式
|
||||
@@ -203,7 +203,11 @@ colibrì 最初是由一人使用 12 核心、25 GB RAM 的筆電開發;
|
||||
```
|
||||
Makefile 根目錄建置/檢查入口
|
||||
c/
|
||||
├── glm.c 單檔 GLM 引擎
|
||||
├── colibri.c GLM 引擎主檔
|
||||
├── quant.h 量化 matmul kernel
|
||||
├── sample.h 取樣與 stop-set
|
||||
├── kv_persist.h .coli_kv 磁碟持久化
|
||||
├── telemetry.h 儀表板協定、統計
|
||||
├── st.h, tok.h, json.h 執行階段標頭檔
|
||||
├── backend_cuda.* 選用的 CUDA 層級
|
||||
├── Makefile 建置與本機檢查
|
||||
@@ -218,7 +222,7 @@ desktop/ 包裝網頁 UI 的 Tauri v2 桌面 shell
|
||||
docs/ 參考文件、實驗與媒體檔
|
||||
```
|
||||
|
||||
執行階段路徑刻意維持扁平、易讀:`glm.c` 加上少量標頭檔。
|
||||
執行階段路徑刻意維持扁平、易讀:`colibri.c` 加上模組化標頭檔。
|
||||
在儲存庫根目錄執行 `make`、`make check` 與 `make clean`,
|
||||
都會轉交給引擎的 Makefile。
|
||||
|
||||
|
||||
+58
-28
@@ -56,11 +56,16 @@ OMPL =
|
||||
endif
|
||||
CFLAGS = -O3 $(OMPC) -Wall -Wextra -Wno-unused-parameter -Wno-misleading-indentation -Wno-unused-function
|
||||
# Opt-in: ARCH=native appends -mcpu=native (arm64 clang uses -mcpu, not -march),
|
||||
# which unlocks the i8mm SMMLA int8/int4 dot kernels in glm.c. ARCH unset ->
|
||||
# which unlocks the i8mm SMMLA int8/int4 dot kernels in colibri.c. ARCH unset ->
|
||||
# no -mcpu, default build byte-identical. Apple clang knows apple-m4 / native.
|
||||
# For older X86_64 Macs (for example, Mac Pro 2019) we need to use -march
|
||||
ifneq ($(ARCH),)
|
||||
ifneq (,$(X86_64))
|
||||
CFLAGS += -march=$(ARCH)
|
||||
else
|
||||
CFLAGS += -mcpu=$(ARCH)
|
||||
endif
|
||||
endif
|
||||
LDFLAGS = -lm $(OMPL)
|
||||
EXE =
|
||||
else ifneq ($(IS_WIN),)
|
||||
@@ -70,7 +75,7 @@ else ifneq ($(IS_WIN),)
|
||||
# ARCH default = x86-64-v3 (portable binary with AVX2). For max speed on THIS
|
||||
# machine use ARCH=native: on AVX-VNNI CPUs (Intel Alder Lake+, Meteor Lake+)
|
||||
# it also unlocks the 128-bit VPDPBUSD int8/int4 dot kernel (dot_i8i8/dot_i4i8),
|
||||
# which the x86-64-v3 baseline does not define. The #ifdef guards in glm.c mean
|
||||
# which the x86-64-v3 baseline does not define. The #ifdef guards in colibri.c mean
|
||||
# a v3 build simply compiles out the VNNI path - safe on any x86-64.
|
||||
CC = gcc
|
||||
ARCH ?= x86-64-v3
|
||||
@@ -207,17 +212,18 @@ LDFLAGS += -framework Metal -framework Foundation -lc++
|
||||
METAL_OBJ = backend_metal.o
|
||||
endif
|
||||
|
||||
all: glm$(EXE)
|
||||
all: colibri$(EXE)
|
||||
|
||||
# phony 'glm' → 'glm.exe' on Windows (so 'make glm' and 'coli build' work on every platform)
|
||||
glm: glm$(EXE)
|
||||
# phony targets — 'glm' kept for backward compatibility
|
||||
colibri: colibri$(EXE)
|
||||
glm: colibri$(EXE)
|
||||
|
||||
# Config stamp: make only tracks file timestamps, not flag changes. Without this,
|
||||
# `make glm.exe CUDA_DLL=1` after a prior CPU-only build reports "up to date" and
|
||||
# silently keeps the CPU-only binary (no CUDA loader) — a build that looks like it
|
||||
# worked but isn't. We record the build-affecting flags in .build-config and rewrite
|
||||
# it ONLY when they change (evaluated here at parse time, so the file's timestamp
|
||||
# moves exactly when the config moves). glm.exe and the CUDA/loader objects depend
|
||||
# `make colibri.exe CUDA_DLL=1` after a prior CPU-only build reports "up to date"
|
||||
# and silently keeps the CPU-only binary (no CUDA loader) — a build that looks like
|
||||
# it worked but isn't. We record the build-affecting flags in .build-config and
|
||||
# rewrite it ONLY when they change (evaluated here at parse time, so the file's
|
||||
# timestamp moves exactly when the config moves). The binary and CUDA/loader objects depend
|
||||
# on it, so they relink on a config change and stay put otherwise. (#306)
|
||||
BUILD_CONFIG := $(CC)|$(CFLAGS)|$(LDFLAGS)|CUDA=$(CUDA)|CUDA_DLL=$(CUDA_DLL)|ARCH=$(ARCH)|CUDA_ARCH=$(CUDA_ARCH)|METAL=$(METAL)
|
||||
BUILD_CONFIG_OLD := $(shell cat .build-config 2>/dev/null)
|
||||
@@ -226,8 +232,8 @@ $(shell printf '%s' '$(BUILD_CONFIG)' > .build-config)
|
||||
endif
|
||||
.build-config: ;
|
||||
|
||||
glm$(EXE): glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h $(CUDA_OBJ) $(METAL_OBJ) .build-config
|
||||
$(CC) $(CFLAGS) glm.c $(CUDA_OBJ) $(METAL_OBJ) -o glm$(EXE) $(LDFLAGS)
|
||||
colibri$(EXE): colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h sample.h kv_persist.h telemetry.h $(CUDA_OBJ) $(METAL_OBJ) .build-config
|
||||
$(CC) $(CFLAGS) colibri.c $(CUDA_OBJ) $(METAL_OBJ) -o colibri$(EXE) $(LDFLAGS)
|
||||
|
||||
# Windows runtime loader object: resolves coli_cuda_* from coli_cuda.dll.
|
||||
backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config
|
||||
@@ -272,8 +278,9 @@ olmoe$(EXE): olmoe.c st.h json.h compat.h
|
||||
# Use a baseline that matches the compiler target. macOS already targets a
|
||||
# portable baseline when ARCH is empty; forcing the x86 value there breaks
|
||||
# Apple Silicon. Unknown targets use native rather than an invalid x86 flag.
|
||||
# Intel Macs need -march for vector instructions
|
||||
ifneq (,$(DARWIN))
|
||||
PORTABLE_ARCH =
|
||||
PORTABLE_ARCH = $(if $(X86_64),x86-64-v3,)
|
||||
else ifneq (,$(AARCH64))
|
||||
PORTABLE_ARCH = armv8-a
|
||||
else ifneq (,$(PPC64))
|
||||
@@ -285,7 +292,7 @@ PORTABLE_ARCH = native
|
||||
endif
|
||||
|
||||
portable:
|
||||
$(MAKE) glm$(EXE) ARCH=$(PORTABLE_ARCH)
|
||||
$(MAKE) colibri$(EXE) ARCH=$(PORTABLE_ARCH)
|
||||
|
||||
iobench$(EXE): iobench.c compat.h
|
||||
$(CC) $(CFLAGS) iobench.c -o iobench$(EXE) $(LDFLAGS)
|
||||
@@ -311,30 +318,30 @@ tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h j
|
||||
tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_idot$(EXE): tests/test_idot.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_idot$(EXE): tests/test_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_stops$(EXE): tests/test_stops.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_stops$(EXE): tests/test_stops.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_topp$(EXE): tests/test_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_topp$(EXE): tests/test_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
# bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test
|
||||
# gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp
|
||||
tests/bench_topp$(EXE): tests/bench_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/bench_topp$(EXE): tests/bench_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_sample_nan$(EXE): tests/test_sample_nan.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_logit_nan$(EXE): tests/test_logit_nan.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_logit_nan$(EXE): tests/test_logit_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
|
||||
@@ -343,15 +350,15 @@ tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
|
||||
tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_dsa_select$(EXE): tests/test_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_dsa_select$(EXE): tests/test_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
# bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356),
|
||||
# NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select
|
||||
tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
tests/test_uring$(EXE): tests/test_uring.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
|
||||
tests/test_uring$(EXE): tests/test_uring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
|
||||
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
|
||||
|
||||
test-c: $(TEST_BINS)
|
||||
@@ -362,18 +369,41 @@ test-python:
|
||||
|
||||
test: test-c test-python
|
||||
|
||||
# --- Efficiency / regression suite (issue: "test the program for inefficiencies") ---
|
||||
# The tiny-model assertions live in test_inefficiency.py and run as part of
|
||||
# test-python (they're discovered by the test_*.py glob). These targets are
|
||||
# convenience entry points; the opt-in full-model report is NEVER in `make test`.
|
||||
#
|
||||
# make efficiency tiny-model asserted regression tests (CPU; part of test-python)
|
||||
# make efficiency-cuda the CUDA-path tests (requires a CUDA build — see below)
|
||||
# make efficiency-report opt-in full-model 🟢/🔴 diagnostic, never fails CI
|
||||
#
|
||||
# CUDA build (Windows): the CUDA tests need a host built with -DCOLI_CUDA plus
|
||||
# the runtime DLL. Do this FIRST — note CUDA_DLL=1 on BOTH the host and the
|
||||
# rule below, or `make glm.exe` will rebuild a CPU-only host and overwrite it:
|
||||
# make clean && make glm.exe CUDA_DLL=1 && make cuda-dll
|
||||
# The tests auto-skip with a clear message if the host is CPU-only.
|
||||
efficiency: test-python
|
||||
$(PYTHON) -m unittest tests.test_inefficiency -v
|
||||
|
||||
efficiency-cuda:
|
||||
$(PYTHON) -m unittest tests.test_inefficiency.TinyCudaEfficiencyTest -v
|
||||
|
||||
efficiency-report:
|
||||
$(PYTHON) tests/test_efficiency_report.py
|
||||
|
||||
# Local validation: one portable CPU build and dependency-free tests.
|
||||
check:
|
||||
$(MAKE) clean
|
||||
$(MAKE) portable
|
||||
$(MAKE) test
|
||||
|
||||
install: glm$(EXE) olmoe$(EXE)
|
||||
install: colibri$(EXE) olmoe$(EXE)
|
||||
$(INSTALL) -d $(DESTDIR)$(BINDIR)
|
||||
$(INSTALL) -d $(DESTDIR)$(LIBEXECDIR)
|
||||
$(INSTALL) -d $(DESTDIR)$(LIBEXECDIR)/tools
|
||||
$(INSTALL) -m 755 coli $(DESTDIR)$(BINDIR)/coli
|
||||
$(INSTALL) -m 755 glm$(EXE) $(DESTDIR)$(LIBEXECDIR)/glm$(EXE)
|
||||
$(INSTALL) -m 755 colibri$(EXE) $(DESTDIR)$(LIBEXECDIR)/colibri$(EXE)
|
||||
$(INSTALL) -m 755 olmoe$(EXE) $(DESTDIR)$(LIBEXECDIR)/olmoe$(EXE)
|
||||
$(INSTALL) -m 644 resource_plan.py doctor.py openai_server.py $(DESTDIR)$(LIBEXECDIR)/
|
||||
$(INSTALL) -m 644 tools/*.py $(DESTDIR)$(LIBEXECDIR)/tools/
|
||||
@@ -387,4 +417,4 @@ clean:
|
||||
|
||||
bench: iobench$(EXE)
|
||||
@if [ -n "$(ARGS)" ]; then ./iobench$(EXE) $(ARGS); else echo "built iobench$(EXE) — run: ./iobench$(EXE) <file> <MB> <iters> <threads> <direct 0|1>"; fi
|
||||
.PHONY: all glm cuda-test cuda-bench cuda-dll portable test-c test-python test check clean install uninstall bench
|
||||
.PHONY: all colibri glm cuda-test cuda-bench cuda-dll portable test-c test-python test check clean install uninstall bench
|
||||
|
||||
@@ -54,16 +54,20 @@ from version import __version__ as _version
|
||||
# guess is right (e.g. a custom packaging layout).
|
||||
_EXE = ".exe" if sys.platform == "win32" else ""
|
||||
_LIBEXEC = os.path.join(os.path.dirname(HERE), "libexec", "colibri")
|
||||
_here_colibri = os.path.join(HERE, "colibri" + _EXE)
|
||||
_here_glm = os.path.join(HERE, "glm" + _EXE)
|
||||
|
||||
if os.environ.get("COLI_ENGINE"):
|
||||
GLM = os.environ["COLI_ENGINE"]
|
||||
TOOLS = os.path.join(os.path.dirname(GLM), "tools")
|
||||
elif os.path.exists(_here_colibri):
|
||||
GLM = _here_colibri
|
||||
TOOLS = os.path.join(HERE, "tools")
|
||||
elif os.path.exists(_here_glm):
|
||||
GLM = _here_glm
|
||||
TOOLS = os.path.join(HERE, "tools")
|
||||
else:
|
||||
GLM = os.path.join(_LIBEXEC, "glm" + _EXE)
|
||||
GLM = os.path.join(_LIBEXEC, "colibri" + _EXE)
|
||||
TOOLS = os.path.join(_LIBEXEC, "tools")
|
||||
sys.path.insert(0, _LIBEXEC) # so `import resource_plan`, `doctor`, `openai_server` still resolve
|
||||
|
||||
@@ -241,13 +245,13 @@ def env_for(a):
|
||||
e["COLI_CUDA"]="0"; e.pop("CUDA_EXPERT_GB",None); e.pop("CUDA_DENSE",None)
|
||||
else:
|
||||
if not cuda_binary():
|
||||
sys.exit(f"{C.yel}--gpu needs the CUDA build:{C.r} make glm CUDA=1 (this binary is CPU-only)")
|
||||
sys.exit(f"{C.yel}--gpu needs the CUDA build:{C.r} make colibri CUDA=1 (this binary is CPU-only)")
|
||||
e["COLI_CUDA"]="1"
|
||||
if a.gpu!="auto": e["COLI_GPUS"]=a.gpu
|
||||
e.setdefault("CUDA_DENSE","1")
|
||||
if a.vram and a.gpu!="none":
|
||||
if not cuda_binary():
|
||||
sys.exit(f"{C.yel}--vram needs the CUDA build:{C.r} make glm CUDA=1 (this binary is CPU-only)")
|
||||
sys.exit(f"{C.yel}--vram needs the CUDA build:{C.r} make colibri CUDA=1 (this binary is CPU-only)")
|
||||
e["COLI_CUDA"]="1"; e["CUDA_EXPERT_GB"]=str(a.vram)
|
||||
return e
|
||||
|
||||
@@ -421,8 +425,8 @@ def cmd_build(a):
|
||||
banner("build")
|
||||
if not os.path.exists(os.path.join(HERE, "Makefile")):
|
||||
sys.exit(f"{C.yel}coli build{C.r} only works from a source checkout (this is an installed copy).\n"
|
||||
f" Clone https://github.com/JustVugg/colibri and run ./setup.sh, or make -C c glm.")
|
||||
sys.exit(subprocess.call(["make","-C",HERE,"glm"]))
|
||||
f" Clone https://github.com/JustVugg/colibri and run ./setup.sh, or make -C c colibri.")
|
||||
sys.exit(subprocess.call(["make","-C",HERE,"colibri"]))
|
||||
|
||||
def cmd_info(a):
|
||||
banner("info")
|
||||
@@ -802,7 +806,7 @@ def cmd_stop(a):
|
||||
if "coli" in cmd and " serve" in cmd and pid!=os.getpid():
|
||||
if not any(p==pid for p,_ in targets): targets.append((pid,"coli serve (cmdline)"))
|
||||
comm=open(f"/proc/{pd}/comm").read().strip()
|
||||
if comm in ("glm","exe","olmoe"):
|
||||
if comm in ("colibri","glm","exe","olmoe"):
|
||||
env=open(f"/proc/{pd}/environ","rb").read().replace(b"\0",b"\n").decode("utf-8","replace")
|
||||
if "SERVE=1" in env: targets.append((pid,f"engine `{comm}` (SERVE=1)"))
|
||||
except (OSError,PermissionError): continue
|
||||
|
||||
+345
-1264
File diff suppressed because it is too large
Load Diff
+121
@@ -0,0 +1,121 @@
|
||||
/* kv_persist.h — .coli_kv on-disk KV cache persistence.
|
||||
* Conversations reopen warm across engine restarts: the compressed MLA KV-cache
|
||||
* is appended incrementally after every turn, crash-safe (nrec written last).
|
||||
* Include after Model/KVState/Cfg are defined; requires now_s() and g_draft. */
|
||||
#ifndef KV_PERSIST_H
|
||||
#define KV_PERSIST_H
|
||||
|
||||
static int g_kvsave=1;
|
||||
#define KV_MAGIC "COLIKV1\0"
|
||||
|
||||
static void kv_hdr(Model *m, int32_t *h, int nrec){
|
||||
Cfg *c=&m->c; int nic=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->Ic && m->Ic[i]) nic++;
|
||||
h[0]=c->n_layers; h[1]=c->kv_lora; h[2]=c->qk_rope;
|
||||
h[3]=m->has_dsa?c->index_hd:0; h[4]=nic; h[5]=c->vocab; h[6]=nrec; h[7]=0;
|
||||
}
|
||||
|
||||
static int64_t kv_rec_bytes(Model *m){
|
||||
Cfg *c=&m->c;
|
||||
int64_t rec = 4 + (int64_t)c->n_layers*(c->kv_lora+c->qk_rope)*4;
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i]) rec+=(int64_t)c->index_hd*4;
|
||||
return rec;
|
||||
}
|
||||
|
||||
static int kv_disk_open(Model *m){
|
||||
KVState *k=m->kv;
|
||||
if(k->disk_fp) return 1;
|
||||
k->disk_fp=fopen(k->disk_path,"r+b");
|
||||
if(!k->disk_fp){
|
||||
k->disk_fp=fopen(k->disk_path,"wb");
|
||||
if(!k->disk_fp) return 0;
|
||||
int32_t h[8]; kv_hdr(m,h,0);
|
||||
fwrite(KV_MAGIC,1,8,k->disk_fp); fwrite(h,4,8,k->disk_fp);
|
||||
fflush(k->disk_fp);
|
||||
fclose(k->disk_fp);
|
||||
k->disk_fp=fopen(k->disk_path,"r+b");
|
||||
if(!k->disk_fp) return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static void kv_disk_truncate(Model *m, int nrec){
|
||||
if(!g_kvsave) return;
|
||||
KVState *k=m->kv;
|
||||
if(k->disk_fp){ fclose(k->disk_fp); k->disk_fp=NULL; }
|
||||
FILE *f=fopen(k->disk_path,"r+b");
|
||||
if(!f){ k->disk_nrec=0; return; }
|
||||
k->disk_nrec=nrec;
|
||||
int32_t nr=nrec; fseek(f,8+6*4,SEEK_SET); fwrite(&nr,4,1,f);
|
||||
fflush(f); fclose(f);
|
||||
}
|
||||
|
||||
static void kv_disk_reset(Model *m){ kv_disk_truncate(m,0); }
|
||||
|
||||
static void kv_disk_append(Model *m, const int *hist, int len){
|
||||
KVState *k=m->kv;
|
||||
if(!g_kvsave || len<=k->disk_nrec) return;
|
||||
Cfg *c=&m->c;
|
||||
if(!kv_disk_open(m)) return;
|
||||
FILE *f=k->disk_fp;
|
||||
int64_t rec = kv_rec_bytes(m);
|
||||
if(rec > k->disk_buf_cap){
|
||||
uint8_t *nb=realloc(k->disk_buf, rec);
|
||||
if(!nb) return;
|
||||
k->disk_buf=nb; k->disk_buf_cap=rec;
|
||||
}
|
||||
fseek(f, 8+8*4 + (int64_t)k->disk_nrec*rec, SEEK_SET);
|
||||
for(int p=k->disk_nrec;p<len;p++){
|
||||
uint8_t *b=k->disk_buf;
|
||||
*(int32_t*)b = hist[p]; b+=4;
|
||||
for(int i=0;i<c->n_layers;i++){
|
||||
memcpy(b, m->Lc[i]+(int64_t)p*c->kv_lora, (size_t)c->kv_lora*4); b+=c->kv_lora*4;
|
||||
memcpy(b, m->Rc[i]+(int64_t)p*c->qk_rope,(size_t)c->qk_rope*4); b+=c->qk_rope*4;
|
||||
}
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i]){
|
||||
memcpy(b, m->Ic[i]+(int64_t)p*c->index_hd, (size_t)c->index_hd*4); b+=c->index_hd*4;
|
||||
}
|
||||
fwrite(k->disk_buf, 1, (size_t)rec, f);
|
||||
}
|
||||
fflush(f);
|
||||
int32_t nr=len; fseek(f,8+6*4,SEEK_SET); fwrite(&nr,4,1,f);
|
||||
fflush(f);
|
||||
k->disk_nrec=len;
|
||||
}
|
||||
|
||||
static int kv_disk_load(Model *m, int *hist, int maxctx){
|
||||
if(!g_kvsave) return 0;
|
||||
KVState *k=m->kv;
|
||||
Cfg *c=&m->c;
|
||||
FILE *f=fopen(k->disk_path,"rb"); if(!f) return 0;
|
||||
char mg[8]; int32_t h[8], w[8]; kv_hdr(m,w,0);
|
||||
if(fread(mg,1,8,f)!=8 || memcmp(mg,KV_MAGIC,8) || fread(h,4,8,f)!=8 ||
|
||||
h[0]!=w[0]||h[1]!=w[1]||h[2]!=w[2]||h[3]!=w[3]||h[4]!=w[4]||h[5]!=w[5]){
|
||||
fprintf(stderr,"[KV] ignoring .coli_kv from a different model or version\n"); fclose(f); return 0; }
|
||||
int nrec=h[6];
|
||||
if(nrec<1){ fclose(f); return 0; }
|
||||
if(nrec>=maxctx-8-g_draft){
|
||||
fprintf(stderr,"[KV] saved conversation (%d tokens) exceeds the context: starting over\n",nrec);
|
||||
fclose(f); return 0; }
|
||||
double t0=now_s();
|
||||
for(int p=0;p<nrec;p++){
|
||||
int32_t tk; if(fread(&tk,4,1,f)!=1){ nrec=p; break; } hist[p]=tk;
|
||||
for(int i=0;i<c->n_layers;i++){
|
||||
if(fread(m->Lc[i]+(int64_t)p*c->kv_lora, 4, c->kv_lora, f)!=(size_t)c->kv_lora ||
|
||||
fread(m->Rc[i]+(int64_t)p*c->qk_rope, 4, c->qk_rope, f)!=(size_t)c->qk_rope){ nrec=p; goto out; }
|
||||
}
|
||||
if(m->has_dsa) for(int i=0;i<c->n_layers;i++) if(m->Ic[i])
|
||||
if(fread(m->Ic[i]+(int64_t)p*c->index_hd, 4, c->index_hd, f)!=(size_t)c->index_hd){ nrec=p; goto out; }
|
||||
}
|
||||
out:
|
||||
fclose(f);
|
||||
if(nrec>0){
|
||||
if(m->has_mtp) m->kv_start[c->n_layers]=-1;
|
||||
fprintf(stderr,"[KV] resumed conversation from disk: %d tokens in %.1fs (no re-prefill)\n",
|
||||
nrec, now_s()-t0);
|
||||
}
|
||||
k->disk_nrec=nrec;
|
||||
return nrec;
|
||||
}
|
||||
|
||||
#endif /* KV_PERSIST_H */
|
||||
@@ -5,6 +5,15 @@
|
||||
* Densa (embed, attn, router, norme, lm_head) residente in RAM (float32).
|
||||
* Expert letti dal disco on-demand via pread+fadvise(DONTNEED), cache LRU per-layer.
|
||||
* Matmul multi-thread con OpenMP (niente BLAS).
|
||||
*
|
||||
* ENV VARS:
|
||||
* PILOT=0/1/2/3 : 0=no prefetch, 1=1-layer lookahead, 2=2-layer, 3=3-layer lookahead
|
||||
* HOT=N : pin top-N hot experts per layer permanently (never evict)
|
||||
* WARMUP=N : tokens before hot pinning activates (default 5)
|
||||
* WIDE=N : prefetch top-K*N candidates (default 1, try 2 or 3)
|
||||
* SMOOTH=F : EMA coefficient for routing momentum (default 0.3, range 0.0-0.95)
|
||||
* CONF_LIMIT=F : cumulative gate probability threshold for prefetch cutoff (default 0.92)
|
||||
* (expert queue is sorted by eid for SSD read locality)
|
||||
*/
|
||||
#define _GNU_SOURCE
|
||||
#include <stdio.h>
|
||||
@@ -12,11 +21,22 @@
|
||||
#include <string.h>
|
||||
#include <math.h>
|
||||
#include <time.h>
|
||||
#include <pthread.h>
|
||||
#if defined(__APPLE__) || defined(__linux__) || defined(__FreeBSD__)
|
||||
#include <sys/resource.h>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include "st.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#define sleep_ms(ms) Sleep(ms)
|
||||
#else
|
||||
#define sleep_ms(ms) usleep((ms) * 1000)
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
/* ---------- config ---------- */
|
||||
typedef struct {
|
||||
int hidden, n_layers, n_heads, n_kv_heads, head_dim;
|
||||
@@ -33,22 +53,61 @@ typedef struct {
|
||||
* Ogni weight [out,in] tenuto come int8 (per-riga) + scala float per riga.
|
||||
* Cosi' la RAM-cache scende da 4 byte/param (f32) a 1 byte/param: e' il
|
||||
* meccanismo che fa stare GLM-5.2 nei 15 GB. dequant-on-use nel matmul. */
|
||||
typedef struct { int eid; int8_t *g, *u, *d; float *gs, *us, *ds; uint64_t used; } Slot;
|
||||
/* pinned=1 means this slot is strongly preferred to keep (hot expert); it will
|
||||
* not be evicted during normal LRU eviction, but may be displaced under extreme
|
||||
* cache pressure when all slots are pinned or in-flight. */
|
||||
typedef struct { int eid; int pinned; int8_t *g, *u, *d; float *gs, *us, *ds; uint64_t used; } Slot;
|
||||
typedef struct { Slot *slots; int n, cap; } LCache;
|
||||
|
||||
typedef struct {
|
||||
Cfg c;
|
||||
shards S;
|
||||
int quant_bits; /* bit di quantizzazione degli expert (2..8); storage int8, niente f32 (#134) */
|
||||
int quant_bits;
|
||||
float *embed, *lm_head, *final_norm;
|
||||
Layer *L;
|
||||
LCache *cache; /* [n_layers] */
|
||||
uint64_t clock, hits, miss;
|
||||
/* kv-cache per-layer: K,V come [H * maxT * head_dim] */
|
||||
float **K, **V; int kv_len, max_t;
|
||||
double dense_load_s;
|
||||
/* IMPROVEMENT 2: expert frequency heatmap */
|
||||
uint32_t *freq;
|
||||
int freq_token_count, hot_pinned, hot_n, warmup_tokens;
|
||||
int token_count;
|
||||
/* PREDICTION IMPROVEMENT A: per-layer EMA of gate logits across tokens.
|
||||
* momentum_logits[l*E .. (l+1)*E-1] = EMA of gate outputs for layer l.
|
||||
* Used exclusively by the PILOT prefetcher to stabilise routing predictions
|
||||
* across tokens; does NOT affect actual MoE routing (pr is unchanged). */
|
||||
float *momentum_logits; /* [n_layers * n_experts], EMA of gate logits */
|
||||
float pilot_smooth; /* SMOOTH env: EMA coefficient 0.0-0.9 (default 0.3) */
|
||||
uint8_t *is_pinned; /* [n_layers * n_experts], 1 if expert is globally pinned */
|
||||
uint8_t *is_queued; /* [n_layers * n_experts], 1 if expert is currently in the prefetch queue */
|
||||
float pilot_conf_limit; /* CONF_LIMIT env: cumulative gate probability threshold (e.g. 0.92) */
|
||||
} Model;
|
||||
|
||||
static pthread_mutex_t g_pilot_mx = PTHREAD_MUTEX_INITIALIZER;
|
||||
static struct { int l, e; } pilot_q[4096];
|
||||
static volatile unsigned pilot_r = 0, pilot_w = 0;
|
||||
static Model *pilot_m = NULL;
|
||||
static int g_pilot = 0;
|
||||
static int g_wide = 1; /* IMPROVEMENT 4: top-K * g_wide candidates prefetched */
|
||||
|
||||
static void pilot_prefetch(Model *m, int lnext, const float *x, int S);
|
||||
static void *pilot_worker(void *arg);
|
||||
static void ensure_pilot_worker_started(Model *m);
|
||||
static void slot_ensure_allocated(Model *m, Slot *s);
|
||||
|
||||
static void ensure_pilot_worker_started(Model *m) {
|
||||
if (!pilot_m) {
|
||||
pilot_m = m;
|
||||
pthread_t t;
|
||||
if (pthread_create(&t, NULL, pilot_worker, NULL) != 0) {
|
||||
fprintf(stderr, "Error: Failed to create pilot prefetch worker thread\n");
|
||||
exit(1);
|
||||
}
|
||||
pthread_detach(t);
|
||||
}
|
||||
}
|
||||
|
||||
/* ---------- utility ---------- */
|
||||
static double now_s(void) { struct timespec t; clock_gettime(CLOCK_MONOTONIC, &t); return t.tv_sec + t.tv_nsec*1e-9; }
|
||||
#if defined(__APPLE__)
|
||||
@@ -210,51 +269,224 @@ static void model_init(Model *m, const char *snap, int cap, int bits) {
|
||||
#undef LD
|
||||
}
|
||||
m->cache = calloc(c->n_layers, sizeof(LCache));
|
||||
for (int i = 0; i < c->n_layers; i++) { m->cache[i].cap = cap; m->cache[i].slots = calloc(cap, sizeof(Slot)); }
|
||||
for (int i = 0; i < c->n_layers; i++) {
|
||||
m->cache[i].cap = cap;
|
||||
m->cache[i].slots = calloc(cap, sizeof(Slot));
|
||||
}
|
||||
/* IMPROVEMENT 2: frequency heatmap for hot expert pinning */
|
||||
m->freq = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint32_t));
|
||||
m->hot_pinned = 0; m->freq_token_count = 0;
|
||||
m->hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0;
|
||||
m->warmup_tokens = getenv("WARMUP") ? atoi(getenv("WARMUP")) : 5;
|
||||
m->token_count = 0;
|
||||
/* PREDICTION A: routing momentum — EMA of gate logits across tokens.
|
||||
* Initialized to zero; first token sets EMA = fresh logits. */
|
||||
m->momentum_logits = calloc((size_t)c->n_layers * c->n_experts, sizeof(float));
|
||||
float sv = getenv("SMOOTH") ? (float)atof(getenv("SMOOTH")) : 0.3f;
|
||||
if (sv < 0.f) sv = 0.f; if (sv > 0.95f) sv = 0.95f;
|
||||
m->pilot_smooth = sv;
|
||||
m->is_pinned = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint8_t));
|
||||
m->is_queued = calloc((size_t)c->n_layers * c->n_experts, sizeof(uint8_t));
|
||||
float cl = getenv("CONF_LIMIT") ? (float)atof(getenv("CONF_LIMIT")) : 0.92f;
|
||||
if (cl < 0.1f) cl = 0.1f; if (cl > 1.0f) cl = 1.0f;
|
||||
m->pilot_conf_limit = cl;
|
||||
m->dense_load_s = now_s() - t0;
|
||||
|
||||
// Persistent Hot Pinning: try to load hot_pinned.bin
|
||||
char pinpath[512];
|
||||
snprintf(pinpath, sizeof(pinpath), "%s/hot_pinned.bin", snap);
|
||||
FILE *pinf = fopen(pinpath, "rb");
|
||||
if (pinf) {
|
||||
size_t expected_size = (size_t)c->n_layers * c->n_experts;
|
||||
if (fread(m->is_pinned, 1, expected_size, pinf) == expected_size) {
|
||||
m->hot_pinned = 1;
|
||||
printf("[HOT] Loaded persistent pinning from %s\n", pinpath);
|
||||
|
||||
if (g_pilot) {
|
||||
ensure_pilot_worker_started(m);
|
||||
for (int l = 0; l < c->n_layers; l++) {
|
||||
for (int e = 0; e < c->n_experts; e++) {
|
||||
if (m->is_pinned[l * c->n_experts + e]) {
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
if (w - r < 4096) {
|
||||
pilot_q[w & 4095].l = l; pilot_q[w & 4095].e = e;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
m->is_queued[l * c->n_experts + e] = 1;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
__atomic_store_n(&pilot_w, w + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
printf("[HOT] Pre-loading pinned experts into cache...\n");
|
||||
double t_wait = now_s();
|
||||
while (1) {
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_ACQUIRE);
|
||||
if (r == w) break;
|
||||
sleep_ms(2);
|
||||
}
|
||||
printf("[HOT] Pre-loaded in %.1fs!\n", now_s() - t_wait);
|
||||
}
|
||||
}
|
||||
fclose(pinf);
|
||||
}
|
||||
}
|
||||
|
||||
/* legge un weight dal disco (streaming) e lo quantizza in q[O,I]+scale[O].
|
||||
* Container pre-quantizzato (convert_olmoe.py: int8 + scale f32 in "name.qs"):
|
||||
* lettura raw diretta — meta' I/O e zero quantize_rows a runtime. Prima di
|
||||
* questa patch il container int8 causava SIGBUS (st_read_f32 su tensori I8). */
|
||||
static void load_expert_w(Model *m, const char *name, int8_t *q, float *scale, int O, int I, float *tmp) {
|
||||
st_tensor *t = st_find(&m->S, name);
|
||||
if (t && t->dtype == 3) { /* I8/U8: container colibri */
|
||||
char qs[300]; snprintf(qs, sizeof(qs), "%s.qs", name);
|
||||
st_read_raw(&m->S, name, q, 1);
|
||||
st_read_f32(&m->S, qs, scale, 1);
|
||||
return;
|
||||
static void slot_ensure_allocated(Model *m, Slot *s) {
|
||||
if (s->g) return;
|
||||
Cfg *c = &m->c;
|
||||
int64_t ng = (int64_t)c->inter * c->hidden;
|
||||
int64_t nd = (int64_t)c->hidden * c->inter;
|
||||
int8_t *w_block = malloc(ng + ng + nd);
|
||||
if (!w_block) {
|
||||
fprintf(stderr, "Error: Out of memory allocating slot weights block\n");
|
||||
exit(1);
|
||||
}
|
||||
st_read_f32(&m->S, name, tmp, 1); /* pread + fadvise DONTNEED */
|
||||
quantize_rows(tmp, q, scale, O, I, m->quant_bits);
|
||||
s->g = w_block;
|
||||
s->u = w_block + ng;
|
||||
s->d = w_block + ng + ng;
|
||||
float *s_block = falloc(c->inter + c->inter + c->hidden);
|
||||
s->gs = s_block;
|
||||
s->us = s_block + c->inter;
|
||||
s->ds = s_block + c->inter + c->inter;
|
||||
s->pinned = 0;
|
||||
}
|
||||
|
||||
static void load_expert_merged(Model *m, int layer, int eid, Slot *s) {
|
||||
char nm[256], qsnm[256];
|
||||
snprintf(nm, sizeof(nm), "model.layers.%d.mlp.experts.%d.merged_weight", layer, eid);
|
||||
snprintf(qsnm, sizeof(qsnm), "model.layers.%d.mlp.experts.%d.qs", layer, eid);
|
||||
st_read_raw(&m->S, nm, s->g, 1);
|
||||
st_read_f32(&m->S, qsnm, s->gs, 0); /* scales are F32; use typed reader for dtype safety */
|
||||
}
|
||||
|
||||
/* ---------- cache expert: ritorna i pesi quantizzati (q+scale) da cache o disco ---------- */
|
||||
static void expert_get(Model *m, int layer, int eid, Slot **out) {
|
||||
LCache *lc = &m->cache[layer];
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
for (int i = 0; i < lc->n; i++) if (lc->slots[i].eid == eid) {
|
||||
m->hits++; lc->slots[i].used = ++m->clock; *out = &lc->slots[i]; return;
|
||||
m->hits++; lc->slots[i].used = ++m->clock; *out = &lc->slots[i];
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
m->miss++;
|
||||
Cfg *c = &m->c;
|
||||
int64_t ng = (int64_t)c->inter * c->hidden, nd = (int64_t)c->hidden * c->inter;
|
||||
Slot *s;
|
||||
if (lc->n < lc->cap) {
|
||||
s = &lc->slots[lc->n++];
|
||||
s->g = malloc(ng); s->u = malloc(ng); s->d = malloc(nd);
|
||||
s->gs = falloc(c->inter); s->us = falloc(c->inter); s->ds = falloc(c->hidden);
|
||||
} else { int lru = 0; for (int i = 1; i < lc->n; i++) if (lc->slots[i].used < lc->slots[lru].used) lru = i; s = &lc->slots[lru]; }
|
||||
float *tmp = falloc(ng > nd ? ng : nd);
|
||||
char nm[256];
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.gate_proj.weight",layer,eid); load_expert_w(m,nm,s->g,s->gs,c->inter,c->hidden,tmp);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.up_proj.weight", layer,eid); load_expert_w(m,nm,s->u,s->us,c->inter,c->hidden,tmp);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.%d.down_proj.weight",layer,eid); load_expert_w(m,nm,s->d,s->ds,c->hidden,c->inter,tmp);
|
||||
free(tmp);
|
||||
s->eid = eid; s->used = ++m->clock;
|
||||
slot_ensure_allocated(m, s);
|
||||
} else {
|
||||
/* LRU eviction — skip pinned and in-flight (eid==-1) slots */
|
||||
int lru = -1;
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].pinned || lc->slots[i].eid < 0) continue;
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
if (lru < 0) {
|
||||
/* All slots are pinned or in-flight; find oldest non-in-flight slot
|
||||
* (may be pinned, but never select one currently being loaded). */
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid < 0) continue; /* never evict in-flight */
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
}
|
||||
if (lru < 0) lru = 0; /* absolute last resort: all in-flight, evict slot 0 */
|
||||
s = &lc->slots[lru];
|
||||
s->pinned = 0;
|
||||
}
|
||||
s->eid = -1;
|
||||
s->used = ++m->clock;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
load_expert_merged(m, layer, eid, s);
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
s->eid = eid;
|
||||
s->pinned = m->is_pinned[layer * c->n_experts + eid];
|
||||
s->used = ++m->clock;
|
||||
*out = s;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
|
||||
/* ---------- IMPROVEMENT 2: pin top-N hot experts per layer ---------- */
|
||||
static void pin_hot_experts(Model *m) {
|
||||
Cfg *c = &m->c;
|
||||
if (m->hot_n <= 0 || m->hot_pinned) return;
|
||||
m->hot_pinned = 1;
|
||||
|
||||
int is_dynamic = (m->hot_n >= 100);
|
||||
double thresh = is_dynamic ? (double)m->hot_n / 1000.0 : 0.0;
|
||||
|
||||
int pinned_total = 0;
|
||||
for (int l = 0; l < c->n_layers; l++) {
|
||||
uint32_t *freq_l = m->freq + (int64_t)l * c->n_experts;
|
||||
|
||||
uint64_t layer_total = 0;
|
||||
for (int e = 0; e < c->n_experts; e++) layer_total += freq_l[e];
|
||||
if (layer_total == 0) continue;
|
||||
|
||||
int max_pin = m->cache[l].cap - 8;
|
||||
if (max_pin < 4) max_pin = 4;
|
||||
|
||||
int hn = is_dynamic ? max_pin : (m->hot_n < c->n_experts ? m->hot_n : c->n_experts);
|
||||
if (hn > 256) hn = 256;
|
||||
int hot_eids[256];
|
||||
int actual_hn = 0;
|
||||
|
||||
for (int k = 0; k < hn; k++) {
|
||||
int best = -1; uint32_t bv = 0;
|
||||
for (int e = 0; e < c->n_experts; e++) {
|
||||
int already = 0;
|
||||
for (int j = 0; j < k; j++) if (hot_eids[j] == e) { already = 1; break; }
|
||||
if (!already && freq_l[e] > bv) { bv = freq_l[e]; best = e; }
|
||||
}
|
||||
if (best < 0 || bv == 0) break;
|
||||
if (is_dynamic && bv < thresh * layer_total) break;
|
||||
hot_eids[k] = best;
|
||||
actual_hn++;
|
||||
}
|
||||
|
||||
for (int k = 0; k < actual_hn; k++) {
|
||||
int eid = hot_eids[k];
|
||||
m->is_pinned[l * c->n_experts + eid] = 1;
|
||||
|
||||
LCache *lc = &m->cache[l];
|
||||
int found = 0;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid == eid) { lc->slots[i].pinned = 1; found = 1; break; }
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
if (!found && g_pilot > 0) {
|
||||
/* Only enqueue when the prefetch worker is active (PILOT>0). */
|
||||
ensure_pilot_worker_started(m);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
int gidx = l * c->n_experts + eid;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
int already = m->is_queued[gidx];
|
||||
if (!already && w - r < 4096) {
|
||||
pilot_q[w & 4095].l = l; pilot_q[w & 4095].e = eid;
|
||||
m->is_queued[gidx] = 1;
|
||||
__atomic_store_n(&pilot_w, w + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
pinned_total++;
|
||||
}
|
||||
}
|
||||
if (is_dynamic) {
|
||||
printf("[HOT] Dynamic Pinned %d experts total (thresh=%.1f%%) after %d warmup tokens\n",
|
||||
pinned_total, thresh * 100.0, m->freq_token_count);
|
||||
} else {
|
||||
printf("[HOT] Pinned %d experts (top-%d/layer) after %d warmup tokens\n",
|
||||
pinned_total, m->hot_n, m->freq_token_count);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* ---------- RoPE su un vettore di una testa (head_dim) a posizione assoluta pos ---------- */
|
||||
static void rope_head(float *x, int pos, const Cfg *c) {
|
||||
int h = c->head_dim / 2;
|
||||
@@ -325,6 +557,19 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
float *g = falloc(I), *u = falloc(I), *hh = falloc(D);
|
||||
for (int s = 0; s < S; s++) {
|
||||
float *pr = logits + (int64_t)s*E;
|
||||
if (m->momentum_logits && m->pilot_smooth > 0.f) {
|
||||
float *ema = m->momentum_logits + (int64_t)layer * E;
|
||||
int is_zero = 1;
|
||||
for (int e = 0; e < E; e++) { if (ema[e] != 0.f) { is_zero = 0; break; } }
|
||||
if (is_zero) {
|
||||
for (int e = 0; e < E; e++) ema[e] = pr[e];
|
||||
} else {
|
||||
for (int e = 0; e < E; e++) {
|
||||
ema[e] = (1.f - m->pilot_smooth) * pr[e] + m->pilot_smooth * ema[e];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
softmax_row(pr, E);
|
||||
/* top-K indici (selezione parziale) */
|
||||
int idx[64]; float val[64];
|
||||
@@ -337,6 +582,11 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
idx[kk] = best; val[kk] = bv;
|
||||
}
|
||||
if (c->norm_topk) { float sm=0; for(int kk=0;kk<K;kk++) sm+=val[kk]; for(int kk=0;kk<K;kk++) val[kk]/=sm; }
|
||||
/* IMPROVEMENT 2: update activation heatmap (before pinning activates) */
|
||||
if (!m->hot_pinned && m->freq) {
|
||||
uint32_t *freq_l = m->freq + (int64_t)layer * E;
|
||||
for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) freq_l[idx[kk]]++;
|
||||
}
|
||||
const float *xs = x + (int64_t)s*D;
|
||||
for (int kk = 0; kk < K; kk++) {
|
||||
Slot *e; expert_get(m, layer, idx[kk], &e);
|
||||
@@ -352,9 +602,20 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
|
||||
free(logits); free(g); free(u); free(hh);
|
||||
}
|
||||
|
||||
/* un passo: token nuovi ids[S] a posizione pos_base. Ritorna logits dell'ultimo token (malloc'd). */
|
||||
static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
Cfg *c = &m->c; int D = c->hidden;
|
||||
if (g_pilot && m->token_count > 0) {
|
||||
/* Flush stale prefetch requests: clear is_queued so pilot_realload
|
||||
* will skip any entries still sitting in pilot_q for the previous
|
||||
* token. We deliberately do NOT move pilot_w backwards; that would
|
||||
* break the ring-buffer invariant (pilot_r could exceed pilot_w if
|
||||
* the worker consumed an entry concurrently). The worker will drain
|
||||
* the stale slots harmlessly because pilot_realload already exits
|
||||
* early when the expert is already cached or is_queued is clear. */
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
memset(m->is_queued, 0, (size_t)c->n_layers * c->n_experts);
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
float *x = falloc((int64_t)S*D);
|
||||
for (int s = 0; s < S; s++) memcpy(x + (int64_t)s*D, m->embed + (int64_t)ids[s]*D, D*sizeof(float));
|
||||
float *nrm = falloc((int64_t)S*D), *tmp = falloc((int64_t)S*D);
|
||||
@@ -363,12 +624,26 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
for (int s = 0; s < S; s++) rmsnorm_row(nrm + (int64_t)s*D, x + (int64_t)s*D, l->in_ln, D, c->eps);
|
||||
attention(m, l, i, nrm, S, pos_base, tmp);
|
||||
for (int64_t j = 0; j < (int64_t)S*D; j++) x[j] += tmp[j];
|
||||
/* IMPROVEMENT 1: PILOT=1 -> 1-layer lookahead */
|
||||
if (g_pilot >= 1 && S <= 8 && i + 1 < c->n_layers)
|
||||
pilot_prefetch(m, i + 1, x, S);
|
||||
for (int s = 0; s < S; s++) rmsnorm_row(nrm + (int64_t)s*D, x + (int64_t)s*D, l->post_ln, D, c->eps);
|
||||
moe(m, l, i, nrm, S, tmp);
|
||||
for (int64_t j = 0; j < (int64_t)S*D; j++) x[j] += tmp[j];
|
||||
|
||||
/* PREDICTION IMPROVEMENT C (Residual gate trick):
|
||||
* PILOT=2 -> prefetch layer i+2 using completed state x (containing MoE residual). */
|
||||
if (g_pilot >= 2 && S <= 8 && i + 2 < c->n_layers)
|
||||
pilot_prefetch(m, i + 2, x, S);
|
||||
if (g_pilot >= 3 && S <= 8 && i + 3 < c->n_layers)
|
||||
pilot_prefetch(m, i + 3, x, S);
|
||||
|
||||
}
|
||||
/* count actual tokens processed (S>1 during prefill) */
|
||||
m->token_count += S; m->freq_token_count += S;
|
||||
if (!m->hot_pinned && m->hot_n > 0 && m->freq_token_count >= m->warmup_tokens)
|
||||
pin_hot_experts(m);
|
||||
m->kv_len = pos_base + S;
|
||||
/* solo l'ultimo token -> logits */
|
||||
float *last = falloc(D);
|
||||
rmsnorm_row(last, x + (int64_t)(S-1)*D, m->final_norm, D, c->eps);
|
||||
float *logit = falloc(c->vocab);
|
||||
@@ -377,6 +652,192 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
|
||||
return logit;
|
||||
}
|
||||
|
||||
static void pilot_realload(Model *m, int layer, int eid) {
|
||||
LCache *lc = &m->cache[layer];
|
||||
Cfg *c = &m->c;
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
/* Early-exit if entry was flushed (is_queued cleared) while waiting. */
|
||||
if (!m->is_queued[layer * c->n_experts + eid]) {
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].eid == eid) {
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return;
|
||||
}
|
||||
}
|
||||
Slot *s;
|
||||
if (lc->n < lc->cap) {
|
||||
s = &lc->slots[lc->n++];
|
||||
slot_ensure_allocated(m, s);
|
||||
} else {
|
||||
/* LRU eviction — skip pinned and in-flight (eid==-1) slots */
|
||||
int lru = -1;
|
||||
for (int i = 0; i < lc->n; i++) {
|
||||
if (lc->slots[i].pinned || lc->slots[i].eid < 0) continue;
|
||||
if (lru < 0 || lc->slots[i].used < lc->slots[lru].used) lru = i;
|
||||
}
|
||||
if (lru < 0) {
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
return; /* all pinned/in-flight, skip */
|
||||
}
|
||||
s = &lc->slots[lru]; s->pinned = 0;
|
||||
}
|
||||
s->eid = -1; s->used = ++m->clock;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
load_expert_merged(m, layer, eid, s);
|
||||
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
s->eid = eid;
|
||||
s->pinned = m->is_pinned[layer * c->n_experts + eid];
|
||||
s->used = ++m->clock;
|
||||
m->is_queued[layer * c->n_experts + eid] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
|
||||
static void *pilot_worker(void *arg) {
|
||||
(void)arg;
|
||||
while (1) {
|
||||
unsigned r = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
unsigned w = __atomic_load_n(&pilot_w, __ATOMIC_ACQUIRE);
|
||||
if (r == w) {
|
||||
sleep_ms(1);
|
||||
continue;
|
||||
}
|
||||
int layer = pilot_q[r & 4095].l;
|
||||
int eid = pilot_q[r & 4095].e;
|
||||
pilot_realload(pilot_m, layer, eid);
|
||||
__atomic_store_n(&pilot_r, r + 1, __ATOMIC_RELEASE);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void pilot_prefetch(Model *m, int lnext, const float *x, int S) {
|
||||
if (lnext < 0 || lnext >= m->c.n_layers) return;
|
||||
Cfg *c = &m->c; int D = c->hidden, E = c->n_experts;
|
||||
ensure_pilot_worker_started(m);
|
||||
float *logits = falloc((int64_t)S * E);
|
||||
Layer *l = &m->L[lnext];
|
||||
|
||||
// PREDICTION IMPROVEMENT B: Apply RMSNorm to x using destination layer's post_ln
|
||||
// This scales inputs to the distribution expected by l->gate.
|
||||
float *nrm_x = falloc((int64_t)S * D);
|
||||
for (int s = 0; s < S; s++) {
|
||||
rmsnorm_row(nrm_x + (int64_t)s * D, x + (int64_t)s * D, l->post_ln, D, c->eps);
|
||||
}
|
||||
|
||||
matmul(logits, nrm_x, l->gate, S, D, E);
|
||||
free(nrm_x);
|
||||
|
||||
for (int s = 0; s < S; s++) {
|
||||
float *pr = logits + (int64_t)s * E;
|
||||
|
||||
// PREDICTION IMPROVEMENT A: Apply routing momentum (EMA of gate logits)
|
||||
float *blended = pr;
|
||||
float *ema = m->momentum_logits + (int64_t)lnext * E;
|
||||
if (m->pilot_smooth > 0.f) {
|
||||
blended = falloc(E);
|
||||
int is_zero = 1;
|
||||
for (int e = 0; e < E; e++) { if (ema[e] != 0.f) { is_zero = 0; break; } }
|
||||
if (is_zero) {
|
||||
for (int e = 0; e < E; e++) {
|
||||
ema[e] = pr[e];
|
||||
blended[e] = pr[e];
|
||||
}
|
||||
} else {
|
||||
for (int e = 0; e < E; e++) {
|
||||
blended[e] = (1.f - m->pilot_smooth) * pr[e] + m->pilot_smooth * ema[e];
|
||||
ema[e] = blended[e]; // update EMA
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int cand = 0;
|
||||
int idx[128];
|
||||
|
||||
float max_logit = -1e30f;
|
||||
for (int e = 0; e < E; e++) { if (blended[e] > max_logit) max_logit = blended[e]; }
|
||||
float *exps = falloc(E);
|
||||
float sum_exps = 0.f;
|
||||
for (int e = 0; e < E; e++) {
|
||||
exps[e] = expf(blended[e] - max_logit);
|
||||
sum_exps += exps[e];
|
||||
}
|
||||
|
||||
float cum_sum = 0.f;
|
||||
int min_cand = c->topk;
|
||||
int max_cand = c->topk * g_wide;
|
||||
if (max_cand < min_cand) max_cand = min_cand;
|
||||
if (max_cand > 128) max_cand = 128; /* idx[] buffer bound */
|
||||
if (max_cand > E) max_cand = E;
|
||||
|
||||
for (int kk = 0; kk < max_cand; kk++) {
|
||||
int best = -1; float bv = -1.f;
|
||||
for (int e = 0; e < E; e++) {
|
||||
int taken = 0; for (int j = 0; j < kk; j++) if (idx[j] == e) { taken=1; break; }
|
||||
if (!taken && exps[e] > bv) { bv = exps[e]; best = e; }
|
||||
}
|
||||
if (best < 0) break;
|
||||
idx[kk] = best;
|
||||
cum_sum += bv;
|
||||
cand++;
|
||||
if (cum_sum >= m->pilot_conf_limit * sum_exps && cand >= min_cand) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(exps);
|
||||
|
||||
if (blended != pr) free(blended);
|
||||
|
||||
/* IMPROVEMENT 5: sort candidates by eid for sequential SSD read locality */
|
||||
for (int a = 0; a < cand-1; a++)
|
||||
for (int b = a+1; b < cand; b++)
|
||||
if (idx[b] >= 0 && (idx[a] < 0 || idx[a] > idx[b])) { int t = idx[a]; idx[a] = idx[b]; idx[b] = t; }
|
||||
|
||||
for (int kk = 0; kk < cand; kk++) {
|
||||
int eid = idx[kk];
|
||||
if (eid < 0) continue;
|
||||
int found = 0;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
LCache *lc = &m->cache[lnext];
|
||||
for (int z = 0; z < lc->n; z++) {
|
||||
if (lc->slots[z].eid == eid) { found = 1; break; }
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
if (!found) {
|
||||
int gidx = lnext * E + eid;
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
int already_queued = m->is_queued[gidx];
|
||||
if (!already_queued) {
|
||||
m->is_queued[gidx] = 1;
|
||||
}
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
|
||||
if (!already_queued) {
|
||||
unsigned w2 = __atomic_load_n(&pilot_w, __ATOMIC_RELAXED);
|
||||
unsigned r2 = __atomic_load_n(&pilot_r, __ATOMIC_ACQUIRE);
|
||||
if (w2 - r2 < 4096) {
|
||||
pilot_q[w2 & 4095].l = lnext;
|
||||
pilot_q[w2 & 4095].e = eid;
|
||||
__atomic_store_n(&pilot_w, w2 + 1, __ATOMIC_RELEASE);
|
||||
} else {
|
||||
pthread_mutex_lock(&g_pilot_mx);
|
||||
m->is_queued[gidx] = 0;
|
||||
pthread_mutex_unlock(&g_pilot_mx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
free(logits);
|
||||
}
|
||||
|
||||
|
||||
/* generazione greedy. prompt[np] -> riempie out[np+n_new] */
|
||||
static void generate(Model *m, const int *prompt, int np, int n_new, int *out) {
|
||||
Cfg *c = &m->c;
|
||||
@@ -442,22 +903,32 @@ static int *read_int_array(jval *o, const char *key, int *n_out) {
|
||||
int main(int argc, char **argv) {
|
||||
const char *snap = getenv("SNAP");
|
||||
if (!snap) { fprintf(stderr, "set SNAP=<snapshot directory>\n"); return 1; }
|
||||
int cap = argc > 1 ? atoi(argv[1]) : 16;
|
||||
int bits = argc > 2 ? atoi(argv[2]) : 8;
|
||||
if (bits < 2 || bits > 8) { /* expert storage is int8_t: bits>8 truncates in quantize_rows (#134). f32 mode is not implemented here — int8 is already token-exact vs the oracle. */
|
||||
fprintf(stderr, "quant_bits must be 2..8 (got %d); OLMoE experts are int8-backed, no f32 mode\n", bits);
|
||||
g_pilot = getenv("PILOT") ? atoi(getenv("PILOT")) : 0;
|
||||
g_wide = getenv("WIDE") ? atoi(getenv("WIDE")) : 1;
|
||||
if (g_wide < 1) g_wide = 1;
|
||||
if (g_wide > 4) g_wide = 4;
|
||||
int hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0;
|
||||
int cap = argc > 1 ? atoi(argv[1]) : 16;
|
||||
int bits = argc > 2 ? atoi(argv[2]) : 8;
|
||||
if (bits < 2 || bits > 8) {
|
||||
fprintf(stderr, "quant_bits must be 2..8 (got %d)\n", bits);
|
||||
return 1;
|
||||
}
|
||||
const char *refpath = argc > 3 ? argv[3] : "ref.json";
|
||||
|
||||
FILE *f = fopen(refpath, "rb"); if(!f){perror(refpath);return 1;}
|
||||
float smooth = getenv("SMOOTH") ? (float)atof(getenv("SMOOTH")) : 0.3f;
|
||||
float conf = getenv("CONF_LIMIT") ? (float)atof(getenv("CONF_LIMIT")) : 0.92f;
|
||||
|
||||
printf("== Streaming C engine v2.2 | cache=%d/layer bits=%d pilot=%d wide=%d hot=%d smooth=%.2f conf=%.2f ==\n",
|
||||
cap, bits, g_pilot, g_wide, hot_n, smooth, conf);
|
||||
|
||||
FILE *f = fopen(refpath, "rb"); if (!f) { perror(refpath); return 1; }
|
||||
fseek(f,0,SEEK_END); long n=ftell(f); fseek(f,0,SEEK_SET);
|
||||
char *buf=malloc(n+1); if(fread(buf,1,n,f)!=(size_t)n){} buf[n]=0; fclose(f);
|
||||
char *buf=malloc(n+1); if (fread(buf,1,n,f)!=(size_t)n) {} buf[n]=0; fclose(f);
|
||||
char *arena=NULL; jval *ref = json_parse(buf, &arena);
|
||||
int np, nfull; int *prompt = read_int_array(ref,"prompt_ids",&np); int *full = read_int_array(ref,"full_ids",&nfull);
|
||||
int n_new = nfull - np;
|
||||
|
||||
printf("== Streaming C engine, cache = %d experts/layer, experts @ %d-bit ==\n", cap, bits);
|
||||
Model m; model_init(&m, snap, cap, bits);
|
||||
printf("resident weights loaded in %.1fs | RSS after load: %.2f GB\n", m.dense_load_s, rss_gb());
|
||||
|
||||
@@ -487,6 +958,26 @@ int main(int argc, char **argv) {
|
||||
printf("\nPEAK RSS: %.2f GB\n", rss_gb());
|
||||
printf("Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0,
|
||||
(unsigned long long)m.hits, (unsigned long long)m.miss);
|
||||
|
||||
|
||||
// Persistent Hot Pinning: save dynamic pinning if newly created
|
||||
if (m.hot_pinned) {
|
||||
char pinpath[512];
|
||||
snprintf(pinpath, sizeof(pinpath), "%s/hot_pinned.bin", snap);
|
||||
FILE *pinf_chk = fopen(pinpath, "rb");
|
||||
if (!pinf_chk) {
|
||||
FILE *pinf_save = fopen(pinpath, "wb");
|
||||
if (pinf_save) {
|
||||
size_t expected_size = (size_t)m.c.n_layers * m.c.n_experts;
|
||||
fwrite(m.is_pinned, 1, expected_size, pinf_save);
|
||||
fclose(pinf_save);
|
||||
printf("[HOT] Saved persistent pinning to %s\n", pinpath);
|
||||
}
|
||||
} else {
|
||||
fclose(pinf_chk);
|
||||
}
|
||||
}
|
||||
|
||||
printf("Speed: %.2f tok/s (%.1fs for %d tokens)\n", n_new/dt, dt, n_new);
|
||||
free(buf); free(arena);
|
||||
return 0;
|
||||
|
||||
+27
-5
@@ -374,6 +374,27 @@ def generation_options(body, limit):
|
||||
if body.get("n", 1) != 1:
|
||||
raise APIError(400, "Colibri currently supports `n=1` only.", "n", "unsupported_value")
|
||||
# `tools`/`functions` are handled by render_chat (declaration) + parse_tool_calls (output).
|
||||
# Validate tools/functions structure early so malformed input fails with a clear error.
|
||||
tools_raw = body.get("tools") or body.get("functions")
|
||||
if tools_raw is not None:
|
||||
if not isinstance(tools_raw, list):
|
||||
raise APIError(400, "`tools` must be a non-empty array.", "tools", "invalid_value")
|
||||
if not tools_raw:
|
||||
raise APIError(400, "`tools` must be a non-empty array.", "tools", "invalid_value")
|
||||
for idx, tool in enumerate(tools_raw):
|
||||
if not isinstance(tool, dict):
|
||||
raise APIError(400, f"Each tool must be an object, got {type(tool).__name__} at index {idx}.",
|
||||
f"tools.{idx}", "invalid_value")
|
||||
fn = tool.get("function", tool) if isinstance(tool, dict) else {}
|
||||
if not isinstance(fn, dict):
|
||||
raise APIError(400, f"Tool function must be an object at index {idx}.",
|
||||
f"tools.{idx}.function", "invalid_value")
|
||||
if not fn.get("name"):
|
||||
raise APIError(400, f"Each tool must have a `name` at index {idx}.",
|
||||
f"tools.{idx}.function.name", "invalid_value")
|
||||
if not isinstance(fn["name"], str):
|
||||
raise APIError(400, f"Tool `name` must be a string at index {idx}.",
|
||||
f"tools.{idx}.function.name", "invalid_value")
|
||||
choice = body.get("tool_choice")
|
||||
if choice is not None:
|
||||
if isinstance(choice, str):
|
||||
@@ -863,7 +884,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def generation(self, body, prompt, request_id, chat):
|
||||
def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=None):
|
||||
# COLI_DEBUG tees the engine transaction to stderr: 1 = decoded output stream only,
|
||||
# 2 = both sides (rendered prompt + output). render_chat already folds prior turns and
|
||||
# tool results into `prompt`, so level 2 is the full conversation the engine saw.
|
||||
@@ -875,8 +896,8 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
sys.stderr.write(f"\n===== PROMPT [{request_id}] =====\n{prompt}\n===== OUTPUT [{request_id}] =====\n")
|
||||
sys.stderr.flush()
|
||||
maximum, temperature, top_p = generation_options(body, self.server.max_tokens)
|
||||
tools = (body.get("tools") or body.get("functions") or None) if chat else None
|
||||
if body.get("tool_choice") == "none":
|
||||
# tools and tool_choice come from chat_completion() already processed/filtered
|
||||
if chat and tool_choice == "none":
|
||||
tools = None # client forbade tools: never surface tool_calls
|
||||
cache_slot = body.get("cache_slot")
|
||||
if (cache_slot is not None and
|
||||
@@ -1078,9 +1099,10 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
if not isinstance(enable_thinking, bool):
|
||||
raise APIError(400, "`enable_thinking` must be a boolean.", "enable_thinking")
|
||||
tools = body.get("tools") or body.get("functions") or None
|
||||
tool_choice = body.get("tool_choice")
|
||||
prompt = render_chat(body.get("messages"), enable_thinking, reasoning_effort, tools,
|
||||
body.get("tool_choice"))
|
||||
self.generation(body, prompt, request_id, True)
|
||||
tool_choice)
|
||||
self.generation(body, prompt, request_id, True, tools, tool_choice)
|
||||
|
||||
def completion(self, body, request_id):
|
||||
prompt = body.get("prompt")
|
||||
|
||||
@@ -0,0 +1,672 @@
|
||||
/* quant.h — quantized matmul kernels (header-only, all functions static).
|
||||
* Multi-architecture SIMD: AVX2 / AVX-512 / AVX-VNNI / ARM NEON / NEON-SDOT /
|
||||
* NEON-i8mm / POWER VSX. Pure compute — no Model or QT dependency. */
|
||||
#ifndef COLI_QUANT_H
|
||||
#define COLI_QUANT_H
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <math.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef _OPENMP
|
||||
#include <omp.h>
|
||||
#endif
|
||||
|
||||
/* ---- SIMD includes -------------------------------------------------------- */
|
||||
#ifdef __AVX2__
|
||||
#include <immintrin.h>
|
||||
static inline float hsum256(__m256 v){
|
||||
__m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1);
|
||||
lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh);
|
||||
sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo);
|
||||
}
|
||||
static inline int hsum256_i32(__m256i v){
|
||||
__m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1);
|
||||
lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo);
|
||||
return _mm_cvtsi128_si32(lo);
|
||||
}
|
||||
#endif
|
||||
#if defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
static inline int hsum128_i32(__m128i v){
|
||||
v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v);
|
||||
}
|
||||
#endif
|
||||
#ifdef __ARM_NEON
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
#ifdef __VSX__
|
||||
#include <altivec.h>
|
||||
#undef vector
|
||||
#undef pixel
|
||||
#undef bool
|
||||
#endif
|
||||
|
||||
/* ---- AVX-512 int4->float accumulator -------------------------------------- */
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
static int g_i4_acc512=1;
|
||||
static inline float dot_i4f_avx512(const uint8_t *w,const float *x,int I){
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m512i b8=_mm512_set1_epi32(8);
|
||||
__m512 acc0=_mm512_setzero_ps(),acc1=_mm512_setzero_ps(); int i=0;
|
||||
for(;i+32<=I;i+=32){ __m128i by=_mm_loadu_si128((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi),n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m512 w0=_mm512_cvtepi32_ps(_mm512_sub_epi32(_mm512_cvtepu8_epi32(n0),b8));
|
||||
__m512 w1=_mm512_cvtepi32_ps(_mm512_sub_epi32(_mm512_cvtepu8_epi32(n1),b8));
|
||||
acc0=_mm512_fmadd_ps(_mm512_loadu_ps(x+i),w0,acc0);
|
||||
acc1=_mm512_fmadd_ps(_mm512_loadu_ps(x+i+16),w1,acc1);
|
||||
}
|
||||
return _mm512_reduce_add_ps(_mm512_add_ps(acc0,acc1));
|
||||
}
|
||||
static int i4_acc512_selftest(void){
|
||||
enum { N=224 }; uint8_t w[(N+1)/2]; float x[N];
|
||||
for(int i=0;i<N;i++){
|
||||
int q=((i*13+5)&15)-8;
|
||||
if(!(i&1)) w[i>>1]=(uint8_t)(q+8);
|
||||
else w[i>>1]|=(uint8_t)((q+8)<<4);
|
||||
x[i]=(float)(((i*29+7)%101)-50)/37.f;
|
||||
}
|
||||
for(int n=32;n<=N;n+=32){
|
||||
float ref=0; for(int i=0;i<n;i++) ref+=x[i]*(float)(((w[i>>1]>>((i&1)*4))&15)-8);
|
||||
float got=dot_i4f_avx512(w,x,n),tol=2e-5f*(1.f+fabsf(ref));
|
||||
if(fabsf(got-ref)>tol){ fprintf(stderr,"AVX512 i4 selftest n=%d: %.9g != %.9g\n",n,got,ref); return 0; }
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W[O,I] f32 ---------------------------------- */
|
||||
static void matmul(float *y, const float *x, const float *W, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const float *w=W+(int64_t)o*I;
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; for(int i=0;i<I;i++) a+=xs[i]*w[i]; y[(int64_t)s*O+o]=a; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int8 per-row + scale[O] ------------------- */
|
||||
static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#ifdef __AVX2__
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+8<=I;i+=8){ __m256i wi=_mm256_cvtepi8_epi32(_mm_loadl_epi64((const __m128i*)(w+i)));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), _mm256_cvtepi32_ps(wi), acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+8<=I;i+=8){ int16x8_t w16=vmovl_s8(vld1_s8(w+i));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w16))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w16)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
for(;i<I;i++) a+=xs[i]*(float)w[i]; y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int4 packed (2/byte) + scale[O] ------------ */
|
||||
static void matmul_i4(float *y, const float *x, const uint8_t *q4, const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
if(g_i4_acc512){ a=dot_i4f_avx512(w,xs,I); i=I&~31; }
|
||||
else {
|
||||
#endif
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m4=vdup_n_u8(0x0F); const int8x8_t b8=vdup_n_s8(8);
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint8x8_t by=vld1_u8(w+(i>>1));
|
||||
uint8x8x2_t z=vzip_u8(vand_u8(by,m4), vshr_n_u8(by,4));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u8(z.val[0]),b8));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u8(z.val[1]),b8));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i+8), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+12), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
}
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t byte=w[i>>1]; int lo=(int)(byte&0xF)-8, hi=(int)(byte>>4)-8;
|
||||
a += xs[i]*(float)lo + xs[i+1]*(float)hi; }
|
||||
if(i<I){ uint8_t byte=w[i>>1]; int lo=(int)(byte&0xF)-8; a += xs[i]*(float)lo; }
|
||||
y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int4 packed + per-GROUP scales (fmt=4) ----- */
|
||||
static void matmul_i4_grouped(float *y, const float *x, const uint8_t *q4, const float *scale,
|
||||
int S, int I, int O, int gs){
|
||||
int rb=(I+1)/2; int ng=(I+gs-1)/gs;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){
|
||||
const uint8_t *w=q4+(int64_t)o*rb;
|
||||
const float *scl=scale+(int64_t)o*ng;
|
||||
for(int s=0;s<S;s++){
|
||||
const float *xs=x+(int64_t)s*I; float a=0;
|
||||
for(int g=0; g*gs<I; g++){
|
||||
int base=g*gs; int glen=gs; if(base+glen>I) glen=I-base;
|
||||
float sc=scl[g];
|
||||
int i=base;
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(; i+16<=base+glen; i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a+=hsum256(acc)*sc;
|
||||
#endif
|
||||
for(; i<base+glen; i+=2){
|
||||
if(i+1<base+glen){ uint8_t byte=w[i>>1];
|
||||
a+=(xs[i]*(float)((int)(byte&0xF)-8)+xs[i+1]*(float)((int)(byte>>4)-8))*sc; }
|
||||
else { uint8_t byte=w[i>>1]; a+=xs[i]*(float)((int)(byte&0xF)-8)*sc; }
|
||||
}
|
||||
}
|
||||
y[(int64_t)s*O+o]=a;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- fused gate+up: one OMP dispatch for both matrices -------------------- */
|
||||
static void matmul_i4_pair(float *yg, float *yu, const float *x,
|
||||
const uint8_t *qg, const float *sg,
|
||||
const uint8_t *qu, const float *su, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int z=0;z<2*O;z++){
|
||||
int o=z<O?z:z-O; const uint8_t *w=(z<O?qg:qu)+(int64_t)o*rb;
|
||||
float a=0; int i=0;
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
if(g_i4_acc512){ a=dot_i4f_avx512(w,x,I); i=I&~31; }
|
||||
else {
|
||||
#endif
|
||||
#ifdef __AVX2__
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi32(8);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_loadl_epi64((const __m128i*)(w+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4),hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i nib=_mm_unpacklo_epi8(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b8));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b8));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(x+i),w0,acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(x+i+8),w1,acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m4=vdup_n_u8(0x0F); const int8x8_t b8=vdup_n_s8(8);
|
||||
float32x4_t ac0=vdupq_n_f32(0),ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint8x8_t by=vld1_u8(w+(i>>1));
|
||||
uint8x8x2_t n=vzip_u8(vand_u8(by,m4),vshr_n_u8(by,4));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u8(n.val[0]),b8));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u8(n.val[1]),b8));
|
||||
ac0=vfmaq_f32(ac0,vld1q_f32(x+i),vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1,vld1q_f32(x+i+4),vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0,vld1q_f32(x+i+8),vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1,vld1q_f32(x+i+12),vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
#if defined(__AVX512F__) && defined(__AVX512BW__)
|
||||
}
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t b=w[i>>1]; a+=x[i]*(float)((b&15)-8)+x[i+1]*(float)((b>>4)-8); }
|
||||
if(i<I) a+=x[i]*(float)((w[i>>1]&15)-8);
|
||||
(z<O?yg:yu)[o]=a*(z<O?sg:su)[o];
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- y[S,O] = x[S,I] @ W^T, W int2 packed (4/byte) + scale[O] ------------ */
|
||||
static void matmul_i2(float *y, const float *x, const uint8_t *q2, const float *scale, int S, int I, int O){
|
||||
int rb=(I+3)/4;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for (int o=0;o<O;o++){ const uint8_t *w=q2+(int64_t)o*rb; float sc=scale[o];
|
||||
for (int s=0;s<S;s++){ const float *xs=x+(int64_t)s*I; float a=0; int i=0;
|
||||
#ifdef __AVX2__
|
||||
const __m128i m2=_mm_set1_epi8(0x03); const __m256i b2=_mm256_set1_epi32(2);
|
||||
__m256 acc=_mm256_setzero_ps();
|
||||
for(;i+16<=I;i+=16){ __m128i by=_mm_cvtsi32_si128(*(const int*)(w+(i>>2)));
|
||||
__m128i p0=_mm_and_si128(by,m2), p1=_mm_and_si128(_mm_srli_epi16(by,2),m2);
|
||||
__m128i p2=_mm_and_si128(_mm_srli_epi16(by,4),m2), p3=_mm_and_si128(_mm_srli_epi16(by,6),m2);
|
||||
__m128i lo=_mm_unpacklo_epi8(p0,p1), hi=_mm_unpacklo_epi8(p2,p3);
|
||||
__m128i nib=_mm_unpacklo_epi16(lo,hi);
|
||||
__m256 w0=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(nib),b2));
|
||||
__m256 w1=_mm256_cvtepi32_ps(_mm256_sub_epi32(_mm256_cvtepu8_epi32(_mm_srli_si128(nib,8)),b2));
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i), w0, acc);
|
||||
acc=_mm256_fmadd_ps(_mm256_loadu_ps(xs+i+8), w1, acc); }
|
||||
a=hsum256(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x8_t m2v=vdup_n_u8(3); const int8x8_t b2v=vdup_n_s8(2);
|
||||
float32x4_t ac0=vdupq_n_f32(0), ac1=vdupq_n_f32(0);
|
||||
for(;i+16<=I;i+=16){ uint32_t wd; memcpy(&wd, w+(i>>2), 4);
|
||||
uint8x8_t by=vreinterpret_u8_u32(vdup_n_u32(wd));
|
||||
uint8x8x2_t z01=vzip_u8(vand_u8(by,m2v), vand_u8(vshr_n_u8(by,2),m2v));
|
||||
uint8x8x2_t z23=vzip_u8(vand_u8(vshr_n_u8(by,4),m2v), vshr_n_u8(by,6));
|
||||
uint16x4x2_t zz=vzip_u16(vreinterpret_u16_u8(z01.val[0]), vreinterpret_u16_u8(z23.val[0]));
|
||||
int16x8_t w0=vmovl_s8(vsub_s8(vreinterpret_s8_u16(zz.val[0]),b2v));
|
||||
int16x8_t w1=vmovl_s8(vsub_s8(vreinterpret_s8_u16(zz.val[1]),b2v));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w0))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+4), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w0))));
|
||||
ac0=vfmaq_f32(ac0, vld1q_f32(xs+i+8), vcvtq_f32_s32(vmovl_s16(vget_low_s16(w1))));
|
||||
ac1=vfmaq_f32(ac1, vld1q_f32(xs+i+12), vcvtq_f32_s32(vmovl_s16(vget_high_s16(w1)))); }
|
||||
a=vaddvq_f32(vaddq_f32(ac0,ac1));
|
||||
#endif
|
||||
for(;i<I;i++){ uint8_t byte=w[i>>2]; int sh=(i&3)*2; a += xs[i]*(float)((int)((byte>>sh)&3)-2); }
|
||||
y[(int64_t)s*O+o]=a*sc; } }
|
||||
}
|
||||
|
||||
/* ---- IDOT: integer dot kernels (int8-quantized activations) --------------- */
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
#define IDOT_KERNEL "avx512-vnni"
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
#define IDOT_KERNEL "avx-vnni"
|
||||
#elif defined(__AVX2__)
|
||||
#define IDOT_KERNEL "avx2"
|
||||
#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
#define IDOT_KERNEL "neon-i8mm"
|
||||
#elif defined(__ARM_NEON)
|
||||
#define IDOT_KERNEL "neon"
|
||||
#elif defined(__VSX__)
|
||||
#define IDOT_KERNEL "vsx"
|
||||
#else
|
||||
#define IDOT_KERNEL "scalar"
|
||||
#endif
|
||||
static int g_idot=1;
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
|
||||
static int g_i4s=1;
|
||||
#elif defined(__VSX__)
|
||||
static int g_i4s=1;
|
||||
#else
|
||||
static int g_i4s=2;
|
||||
#endif
|
||||
|
||||
static inline float qrow_i8(const float *x, int8_t *q, int I){
|
||||
float amax=0; for(int i=0;i<I;i++){ float a=fabsf(x[i]); if(a>amax)amax=a; }
|
||||
float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s;
|
||||
for(int i=0;i<I;i++) q[i]=(int8_t)lrintf(x[i]*inv);
|
||||
return s;
|
||||
}
|
||||
|
||||
/* dot int8*int8 */
|
||||
static inline int32_t dot_i8i8(const int8_t *w, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
__m512i acc=_mm512_setzero_si512();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m512i wv=_mm512_loadu_si512((const void*)(w+i));
|
||||
__m512i xv=_mm512_loadu_si512((const void*)(x+i));
|
||||
__mmask64 neg=_mm512_movepi8_mask(wv);
|
||||
__m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
|
||||
acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=_mm512_reduce_add_epi32(acc);
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
__m128i acc=_mm_setzero_si128();
|
||||
for(;i+16<=I;i+=16){
|
||||
__m128i wv=_mm_loadu_si128((const __m128i*)(w+i));
|
||||
__m128i xv=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i xs=_mm_sign_epi8(xv,wv);
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
#elif defined(__AVX2__)
|
||||
__m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1);
|
||||
for(;i+32<=I;i+=32){
|
||||
__m256i wv=_mm256_loadu_si256((const __m256i*)(w+i));
|
||||
__m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
|
||||
__m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
|
||||
acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
|
||||
}
|
||||
sum=hsum256_i32(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
#if defined(__ARM_FEATURE_DOTPROD)
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
|
||||
for(;i+64<=I;i+=64){
|
||||
a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i));
|
||||
a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16));
|
||||
a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32));
|
||||
a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i));
|
||||
sum=vaddvq_s32(acc);
|
||||
#else
|
||||
int32x4_t acc=vdupq_n_s32(0);
|
||||
for(;i+16<=I;i+=16){
|
||||
int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i);
|
||||
int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv));
|
||||
p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#endif
|
||||
#elif defined(__VSX__)
|
||||
__vector signed int acc=vec_splats(0);
|
||||
const __vector signed char vz=vec_splats((signed char)0);
|
||||
for(;i+16<=I;i+=16){
|
||||
__vector signed char wv=vec_xl(0,(const signed char*)(w+i));
|
||||
__vector signed char xv=vec_xl(0,(const signed char*)(x+i));
|
||||
__vector __bool char neg=vec_cmplt(wv,vz);
|
||||
__vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg);
|
||||
__vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg);
|
||||
acc=vec_msum(xs,wa,acc);
|
||||
}
|
||||
sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
|
||||
#endif
|
||||
for(;i<I;i++) sum+=(int32_t)w[i]*x[i];
|
||||
return sum;
|
||||
}
|
||||
|
||||
/* dot int4(packed)*int8 */
|
||||
static inline int32_t dot_i4i8(const uint8_t *w4, const int8_t *x, int I){
|
||||
int32_t sum=0; int i=0;
|
||||
#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
|
||||
const __m256i m4v=_mm256_set1_epi8(0x0F);
|
||||
const __m512i b8v=_mm512_set1_epi8(8);
|
||||
const __m512i xidx=_mm512_setr_epi64(0,1,4,5,2,3,6,7);
|
||||
__m512i acc=_mm512_setzero_si512();
|
||||
for(;i+64<=I;i+=64){
|
||||
__m256i by=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
|
||||
__m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v);
|
||||
__m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi);
|
||||
__m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v);
|
||||
__m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i)));
|
||||
__mmask64 neg=_mm512_movepi8_mask(wv);
|
||||
__m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
|
||||
acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
|
||||
}
|
||||
sum=_mm512_reduce_add_epi32(acc);
|
||||
#elif defined(__AVXVNNI__) && defined(__AVX2__)
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8);
|
||||
__m128i acc=_mm_setzero_si128();
|
||||
for(;i+32<=I;i+=32){
|
||||
__m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m128i w0=_mm_sub_epi8(n0,b8), w1=_mm_sub_epi8(n1,b8);
|
||||
__m128i x0=_mm_loadu_si128((const __m128i*)(x+i));
|
||||
__m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
|
||||
acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
|
||||
}
|
||||
sum=hsum128_i32(acc);
|
||||
#elif defined(__AVX2__)
|
||||
const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8);
|
||||
const __m256i ones=_mm256_set1_epi16(1);
|
||||
__m256i acc=_mm256_setzero_si256();
|
||||
for(;i+32<=I;i+=32){
|
||||
__m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
|
||||
__m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
|
||||
__m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
|
||||
__m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8);
|
||||
__m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
|
||||
__m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
|
||||
acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
|
||||
}
|
||||
sum=hsum256_i32(acc);
|
||||
#elif defined(__ARM_NEON)
|
||||
const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
|
||||
#if defined(__ARM_FEATURE_DOTPROD)
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
|
||||
for(;i+64<=I;i+=64){
|
||||
uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16);
|
||||
uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4));
|
||||
uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4));
|
||||
a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i));
|
||||
a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16));
|
||||
a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32));
|
||||
a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t by=vld1q_u8(w4+(i>>1));
|
||||
uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
|
||||
acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i));
|
||||
acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16));
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#else
|
||||
int32x4_t acc=vdupq_n_s32(0);
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t by=vld1q_u8(w4+(i>>1));
|
||||
uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
|
||||
int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q);
|
||||
int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q);
|
||||
int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16);
|
||||
int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0));
|
||||
p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1));
|
||||
p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1));
|
||||
acc=vpadalq_s16(acc,p);
|
||||
}
|
||||
sum=vaddvq_s32(acc);
|
||||
#endif
|
||||
#elif defined(__VSX__)
|
||||
const __vector unsigned char m4v=vec_splats((unsigned char)0x0F);
|
||||
const __vector unsigned char sh4=vec_splats((unsigned char)4);
|
||||
const __vector signed char b8v=vec_splats((signed char)8);
|
||||
const __vector signed char vz=vec_splats((signed char)0);
|
||||
__vector signed int acc=vec_splats(0);
|
||||
for(;i+32<=I;i+=32){
|
||||
__vector unsigned char by=vec_xl(0,w4+(i>>1));
|
||||
__vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4);
|
||||
__vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v);
|
||||
__vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v);
|
||||
__vector signed char x0=vec_xl(0,(const signed char*)(x+i));
|
||||
__vector signed char x1=vec_xl(0,(const signed char*)(x+i+16));
|
||||
__vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz);
|
||||
acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0),
|
||||
(__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc);
|
||||
acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1),
|
||||
(__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc);
|
||||
}
|
||||
sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
|
||||
#endif
|
||||
for(;i+1<I;i+=2){ uint8_t b=w4[i>>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; }
|
||||
if(i<I){ uint8_t b=w4[i>>1]; sum+=((int)(b&0xF)-8)*x[i]; }
|
||||
return sum;
|
||||
}
|
||||
|
||||
/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1,
|
||||
int8x16_t xs, int8x16_t xs1){
|
||||
acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)),
|
||||
vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1)));
|
||||
return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)),
|
||||
vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1)));
|
||||
}
|
||||
static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q,
|
||||
const float *scale, int S, int I, int O){
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<(O&~1);o+=2){
|
||||
const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I;
|
||||
float sc0=scale[o], sc1=scale[o+1];
|
||||
for(int s=0;s<(S&~1);s+=2){
|
||||
const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
|
||||
for(;i+64<=I;i+=64){
|
||||
a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16));
|
||||
a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32));
|
||||
a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48));
|
||||
}
|
||||
for(;i+16<=I;i+=16)
|
||||
a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i));
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
|
||||
int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
|
||||
for(;i<I;i++){ int a=wo[i],b=wo1[i],u=xs[i],v=xs1[i];
|
||||
d00+=a*u; d01+=a*v; d10+=b*u; d11+=b*v; }
|
||||
y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
|
||||
y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
|
||||
y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
|
||||
}
|
||||
if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
|
||||
y[(int64_t)s*O+o] =(float)dot_i8i8(wo, xs,I)*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)]=(float)dot_i8i8(wo1,xs,I)*sc1*sx[s]; }
|
||||
}
|
||||
if(O&1){ int o=O-1; const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i8i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
static void matmul_i4_idot_mm(float *y, const int8_t *xq, const float *sx, const uint8_t *q4,
|
||||
const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<(O&~1);o+=2){
|
||||
const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
|
||||
const uint8_t *wo=q4+(int64_t)o*rb, *wo1=q4+(int64_t)(o+1)*rb;
|
||||
float sc0=scale[o], sc1=scale[o+1];
|
||||
for(int s=0;s<(S&~1);s+=2){
|
||||
const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
|
||||
int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
|
||||
for(;i+64<=I;i+=64){
|
||||
uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
|
||||
uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16);
|
||||
uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
|
||||
uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
|
||||
uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4));
|
||||
uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4));
|
||||
a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
|
||||
vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
|
||||
a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q),
|
||||
vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32));
|
||||
a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48));
|
||||
}
|
||||
for(;i+32<=I;i+=32){
|
||||
uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
|
||||
uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
|
||||
uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
|
||||
a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
|
||||
vld1q_s8(xs+i), vld1q_s8(xs1+i));
|
||||
a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
|
||||
vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
|
||||
vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
|
||||
}
|
||||
int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
|
||||
int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
|
||||
int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
|
||||
for(;i+1<I;i+=2){ uint8_t bo=wo[i>>1], bo1=wo1[i>>1];
|
||||
int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8;
|
||||
int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1];
|
||||
d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; }
|
||||
if(i<I){ uint8_t bo=wo[i>>1], bo1=wo1[i>>1];
|
||||
int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8;
|
||||
d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; }
|
||||
y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
|
||||
y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
|
||||
y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
|
||||
}
|
||||
if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
|
||||
y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s];
|
||||
y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; }
|
||||
}
|
||||
if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i4i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
#endif
|
||||
|
||||
/* ---- IDOT dispatch (int8-quantized activations) --------------------------- */
|
||||
static void matmul_q_idot(float *y, const int8_t *xq, const float *sx, const int8_t *q,
|
||||
const float *scale, int S, int I, int O){
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
if(S>=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; }
|
||||
#endif
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const int8_t *w=q+(int64_t)o*I; float sc=scale[o];
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i8i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
static void matmul_i4_idot(float *y, const int8_t *xq, const float *sx, const uint8_t *q4,
|
||||
const float *scale, int S, int I, int O){
|
||||
int rb=(I+1)/2;
|
||||
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
if(S>=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; }
|
||||
#endif
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
|
||||
for(int s=0;s<S;s++) y[(int64_t)s*O+o]=(float)dot_i4i8(w,xq+(int64_t)s*I,I)*sc*sx[s]; }
|
||||
}
|
||||
|
||||
/* ---- per-thread quantization scratch -------------------------------------- */
|
||||
typedef struct { int8_t *xq; size_t xq_cap; float *sx; size_t sx_cap; } QScratch;
|
||||
static _Thread_local QScratch g_qscratch;
|
||||
static void quant_scratch(size_t xn, size_t sn, int8_t **xq, float **sx){
|
||||
if(xn>g_qscratch.xq_cap){
|
||||
int8_t *p=realloc(g_qscratch.xq,xn);
|
||||
if(!p){ fprintf(stderr,"OOM quant scratch\n"); exit(1); }
|
||||
g_qscratch.xq=p; g_qscratch.xq_cap=xn;
|
||||
}
|
||||
if(sn>g_qscratch.sx_cap){
|
||||
float *p=realloc(g_qscratch.sx,sn*sizeof(float));
|
||||
if(!p){ fprintf(stderr,"OOM quant scales\n"); exit(1); }
|
||||
g_qscratch.sx=p; g_qscratch.sx_cap=sn;
|
||||
}
|
||||
*xq=g_qscratch.xq; *sx=g_qscratch.sx;
|
||||
}
|
||||
|
||||
/* ---- f32 -> quantized packing --------------------------------------------- */
|
||||
static void quantize_rows(const float *w, int8_t *q, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
int8_t *qr=q+(int64_t)o*I;
|
||||
for(int i=0;i<I;i++){ int v=(int)lrintf(wr[i]/s); if(v>qmax)v=qmax; if(v<-qmax-1)v=-qmax-1; qr[i]=(int8_t)v; }
|
||||
}
|
||||
}
|
||||
static void pack_int4(const float *w, uint8_t *q4, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1, rb=(I+1)/2;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
uint8_t *qr=q4+(int64_t)o*rb;
|
||||
for(int i=0;i<I;i+=2){
|
||||
int v0=(int)lrintf(wr[i]/s); if(v0>qmax)v0=qmax; if(v0<-8)v0=-8;
|
||||
int v1=0; if(i+1<I){ v1=(int)lrintf(wr[i+1]/s); if(v1>qmax)v1=qmax; if(v1<-8)v1=-8; }
|
||||
qr[i>>1] = (uint8_t)((v0+8) | ((v1+8)<<4));
|
||||
}
|
||||
}
|
||||
}
|
||||
static void pack_int2(const float *w, uint8_t *q2, float *scale, int O, int I, int bits){
|
||||
int qmax=(1<<(bits-1))-1, rb=(I+3)/4;
|
||||
#pragma omp parallel for schedule(static)
|
||||
for(int o=0;o<O;o++){ const float *wr=w+(int64_t)o*I; float amax=0;
|
||||
for(int i=0;i<I;i++){ float a=fabsf(wr[i]); if(a>amax)amax=a; }
|
||||
float s=amax/qmax; if(s<1e-8f)s=1e-8f; scale[o]=s;
|
||||
uint8_t *qr=q2+(int64_t)o*rb;
|
||||
for(int i=0;i<I;i+=4){ uint8_t byte=0;
|
||||
for(int k=0;k<4 && i+k<I;k++){ int v=(int)lrintf(wr[i+k]/s); if(v>qmax)v=qmax; if(v<-2)v=-2; byte|=(uint8_t)((v+2)<<(k*2)); }
|
||||
qr[i>>2]=byte;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif /* COLI_QUANT_H */
|
||||
@@ -0,0 +1,28 @@
|
||||
{
|
||||
"prompt_ids": [
|
||||
510,
|
||||
5347,
|
||||
273,
|
||||
6181,
|
||||
310
|
||||
],
|
||||
"full_ids": [
|
||||
510,
|
||||
5347,
|
||||
273,
|
||||
6181,
|
||||
310,
|
||||
7785,
|
||||
15,
|
||||
187,
|
||||
187,
|
||||
510,
|
||||
3565,
|
||||
3448,
|
||||
273,
|
||||
6181,
|
||||
310,
|
||||
5112,
|
||||
15
|
||||
]
|
||||
}
|
||||
+181
-23
@@ -166,14 +166,33 @@ def discover_gpus():
|
||||
return devices
|
||||
|
||||
|
||||
def _physical_cores_warn(message):
|
||||
"""Visibility for a mis-detected core count: a silent "1" here becomes
|
||||
OMP_NUM_THREADS=1 and pins the whole run to a single core (#325). Emit on
|
||||
stderr so it surfaces in the [PLAN]/[OMP] stream without being swallowed."""
|
||||
print(f"[plan] warning: {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def physical_cpu_count():
|
||||
"""Number of physical CPU cores (not SMT siblings).
|
||||
|
||||
Per-expert matmul regions are tiny and back-to-back; two SMT siblings share
|
||||
one AVX-512 unit and contend, so logical (SMT) counts over-subscribe and
|
||||
hurt throughput. We want true physical cores. A silent 1 here propagates to
|
||||
OMP_NUM_THREADS=1 and pins the run to one core (#325), so every fallback
|
||||
must be visible, never just ``or 1``.
|
||||
"""
|
||||
if sys.platform == "win32":
|
||||
# os.cpu_count() conta i processori logici (SMT): 2 thread/core saturano
|
||||
# le unita' AVX-512 e peggiorano il matmul. Contiamo i core fisici veri
|
||||
# con GetLogicalProcessorInformationEx(RelationProcessorCore).
|
||||
# Contiamo i core fisici veri con GetLogicalProcessorInformationEx
|
||||
# (RelationProcessorCore). Le firme vanno dichiarate: su Python a 64 bit
|
||||
# una WinAPI non dichiarata ritorna c_int (32 bit) e riceve i puntatori
|
||||
# come c_int di default, quindi il probe puo' fallire silenziosamente.
|
||||
try:
|
||||
import ctypes
|
||||
k32 = ctypes.windll.kernel32
|
||||
k32.GetLogicalProcessorInformationEx.argtypes = [
|
||||
ctypes.c_uint, ctypes.c_void_p, ctypes.POINTER(ctypes.c_ulong)]
|
||||
k32.GetLogicalProcessorInformationEx.restype = ctypes.c_int
|
||||
need = ctypes.c_ulong(0)
|
||||
k32.GetLogicalProcessorInformationEx(0, None, ctypes.byref(need))
|
||||
buf = (ctypes.c_char * need.value)()
|
||||
@@ -189,18 +208,69 @@ def physical_cpu_count():
|
||||
off += size
|
||||
if cores:
|
||||
return cores
|
||||
except (OSError, ValueError, AttributeError):
|
||||
pass
|
||||
_physical_cores_warn("GetLogicalProcessorInformationEx returned no cores")
|
||||
except (OSError, ValueError, AttributeError) as error:
|
||||
_physical_cores_warn(f"Windows core probe failed: {error}")
|
||||
try:
|
||||
# Ask lscpu for exactly core,socket and dedupe on (core, socket).
|
||||
# Counting un-deduplicated rows would return logical threads (SMT),
|
||||
# which was the original over-subscription bug. Empty fields ("-")
|
||||
# mark an offline core/socket and fail int() -> skipped.
|
||||
#
|
||||
# Column layout robustness: `lscpu -p=<list>` emits *exactly* the
|
||||
# requested columns (no CPU prefix), while bare `lscpu -p` prepends
|
||||
# CPU. We requested two columns, but take the LAST TWO fields so the
|
||||
# parser stays correct whether or not a CPU column is present
|
||||
# (JustVugg review: the previous fields[1]/fields[2] indexing assumed
|
||||
# a 3-column layout and regressed 2-column output to the logical
|
||||
# count -- the opposite of the fix).
|
||||
result = subprocess.run(["lscpu", "-p=core,socket"], text=True,
|
||||
capture_output=True, check=True, timeout=5)
|
||||
cores = {tuple(map(int, line.split(","))) for line in result.stdout.splitlines()
|
||||
if line and not line.startswith("#")}
|
||||
cores = set()
|
||||
for line in result.stdout.splitlines():
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
fields = line.split(",")
|
||||
if len(fields) < 2:
|
||||
continue
|
||||
try:
|
||||
core, socket = int(fields[-2]), int(fields[-1])
|
||||
except ValueError:
|
||||
continue # "-" for an offline core/socket
|
||||
cores.add((core, socket))
|
||||
if cores:
|
||||
return len(cores)
|
||||
except (OSError, ValueError, subprocess.SubprocessError):
|
||||
pass
|
||||
return os.cpu_count() or 1
|
||||
except (OSError, ValueError, subprocess.SubprocessError) as error:
|
||||
_physical_cores_warn(f"lscpu core probe failed: {error}")
|
||||
logical = os.cpu_count()
|
||||
if not logical:
|
||||
_physical_cores_warn(
|
||||
"could not detect any CPU cores; falling back to 1. "
|
||||
"Set OMP_NUM_THREADS manually to fix single-core decode (#325).")
|
||||
return 1
|
||||
_physical_cores_warn(
|
||||
f"physical-core probes unavailable; using {logical} logical CPUs "
|
||||
f"(SMT may over-subscribe). Set OMP_NUM_THREADS to physical cores if slow.")
|
||||
return logical
|
||||
|
||||
|
||||
def _resolve_physical_cores(physical_cpus):
|
||||
"""Coerce the build_plan() physical-core argument to a sane positive int.
|
||||
|
||||
A None/0/None-ish value reaching here means physical_cpu_count() already
|
||||
warned; clamp to 1 (so the engine always gets a positive team size) but keep
|
||||
that clamp visible rather than silently masking it as the old ``max(1, int())``
|
||||
did (#325)."""
|
||||
try:
|
||||
count = int(physical_cpus or 0)
|
||||
except (TypeError, ValueError):
|
||||
count = 0
|
||||
if count < 1:
|
||||
_physical_cores_warn(
|
||||
"physical core count resolved to 0; defaulting to 1. "
|
||||
"Set OMP_NUM_THREADS to fix single-core decode (#325).")
|
||||
return 1
|
||||
return count
|
||||
|
||||
|
||||
def cpu_socket_count():
|
||||
@@ -219,6 +289,54 @@ def cpu_socket_count():
|
||||
return 1
|
||||
|
||||
|
||||
def _auto_tune(bottleneck_class, projected_hit, gpus, cpu_sockets, plan_has_metal):
|
||||
"""Derive tuning knobs from the bottleneck classification."""
|
||||
tune = {}
|
||||
has_gpu = bool(gpus)
|
||||
n_gpu = len(gpus)
|
||||
|
||||
# MTP: costs more than it saves when compute-bound (#389 measured 42% loss)
|
||||
if bottleneck_class == "compute":
|
||||
tune["DRAFT"] = {"value": "0",
|
||||
"reason": "compute-bound: MTP batch overhead exceeds yield"}
|
||||
elif bottleneck_class == "disk" and projected_hit < 0.90:
|
||||
tune["DRAFT"] = {"value": "0",
|
||||
"reason": "low hit rate: MTP widens expert union, adds disk reads"}
|
||||
# otherwise leave DRAFT unset (engine default: auto)
|
||||
|
||||
# PIPE: resident pipeline mode depends on GPU count
|
||||
if has_gpu and n_gpu == 1:
|
||||
tune["COLI_CUDA_PIPE"] = {"value": "1",
|
||||
"reason": "single GPU: S=1 pipeline gate"}
|
||||
elif has_gpu and n_gpu > 1:
|
||||
tune["COLI_CUDA_PIPE"] = {"value": "2",
|
||||
"reason": "multi-GPU: residual stays on-device across layers"}
|
||||
elif not has_gpu and bottleneck_class == "disk":
|
||||
tune["PIPE"] = {"value": "1",
|
||||
"reason": "overlap disk reads with resident expert compute"}
|
||||
|
||||
# NUMA: selective interleave for GPU hosts, blanket hint for CPU-only
|
||||
if cpu_sockets > 1 and has_gpu:
|
||||
tune["COLI_NUMA"] = {"value": "1",
|
||||
"reason": "multi-socket + GPU: interleave expert slabs, protect DMA buffers"}
|
||||
elif cpu_sockets > 1 and not has_gpu:
|
||||
tune["COLI_NUMA"] = {"value": "1",
|
||||
"reason": "multi-socket CPU-only: interleave expert slabs across nodes"}
|
||||
tune["_numa_hint"] = "numactl --interleave=all may perform better on CPU-only hosts"
|
||||
|
||||
# OMP: kill hot-thread spin when GPU/Metal owns the power budget
|
||||
if plan_has_metal:
|
||||
tune["COLI_NO_OMP_TUNE"] = {"value": "1",
|
||||
"reason": "Metal: OMP spin-wait steals GPU power budget"}
|
||||
|
||||
# PIN: fully resident if RAM allows and no GPU tier competes
|
||||
if projected_hit >= 0.99 and not has_gpu:
|
||||
tune["PIN_GB"] = {"value": "all",
|
||||
"reason": "enough RAM for full expert residency"}
|
||||
|
||||
return tune
|
||||
|
||||
|
||||
POLICIES = {
|
||||
"quality": {"preserve_quantization": True, "preserve_router": True},
|
||||
"balanced": {"preserve_quantization": True, "preserve_router": True},
|
||||
@@ -290,19 +408,35 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
|
||||
if cold_bytes:
|
||||
warnings.append("cold expert misses may reach disk; normal decode speed depends on hit rate")
|
||||
|
||||
total_expert = info["expert_bytes"]
|
||||
resident_expert = hot_bytes + warm_bytes
|
||||
projected_hit = resident_expert / total_expert if total_expert else 1.0
|
||||
|
||||
if cold_bytes:
|
||||
bottleneck = "disk expert misses"
|
||||
elif warm_bytes:
|
||||
bottleneck = "CPU expert compute and RAM bandwidth"
|
||||
bottleneck_class = "disk"
|
||||
elif warm_bytes and gpus:
|
||||
bottleneck = "CPU expert tail and GPU compute"
|
||||
bottleneck_class = "mixed"
|
||||
elif projected_hit >= 0.99:
|
||||
if gpus:
|
||||
bottleneck = "GPU compute and interconnect"
|
||||
else:
|
||||
bottleneck = "CPU expert compute (fully resident)"
|
||||
bottleneck_class = "compute"
|
||||
else:
|
||||
bottleneck = "GPU compute and interconnect"
|
||||
bottleneck = "CPU expert compute and RAM bandwidth"
|
||||
bottleneck_class = "memory"
|
||||
|
||||
tune = _auto_tune(bottleneck_class, projected_hit, gpus, cpu_sockets,
|
||||
plan_has_metal=False)
|
||||
|
||||
return {
|
||||
"version": 2,
|
||||
"policy": {"name": policy, **POLICIES[policy],
|
||||
"quality_preserving": policy != "experimental-fast"},
|
||||
"model": {key: value for key, value in info.items() if key != "config"},
|
||||
"cpu": {"physical_cores": max(1, int(physical_cpus)),
|
||||
"cpu": {"physical_cores": _resolve_physical_cores(physical_cpus),
|
||||
"sockets": max(1, int(cpu_sockets)),
|
||||
"thread_policy": "physical-cores"},
|
||||
"tiers": {
|
||||
@@ -317,6 +451,9 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
|
||||
"expert_capacity": vram_experts, "requires_host_backing": False},
|
||||
},
|
||||
"expected_bottleneck": bottleneck,
|
||||
"bottleneck_class": bottleneck_class,
|
||||
"projected_hit_rate": round(projected_hit, 4),
|
||||
"tune": tune,
|
||||
"decisions": [
|
||||
{"target": "VRAM", "reason": "profile-ranked hot experts"},
|
||||
{"target": "RAM", "reason": "warm experts execute on CPU without quality loss"},
|
||||
@@ -331,15 +468,23 @@ def environment_for_plan(plan, env=None, cuda_enabled=True):
|
||||
result = dict(env or {})
|
||||
result.setdefault("COLI_POLICY", plan["policy"]["name"])
|
||||
result.setdefault("OMP_NUM_THREADS", str(plan["cpu"]["physical_cores"]))
|
||||
if sys.platform != "win32":
|
||||
# la libgomp di MinGW non supporta l'affinity su Windows
|
||||
# ("Affinity not supported on this configuration"): non impostarle li'.
|
||||
result.setdefault("OMP_PROC_BIND", "spread")
|
||||
result.setdefault("OMP_PLACES", "cores")
|
||||
if sys.platform.startswith("linux") and plan["cpu"].get("sockets", 1) > 1:
|
||||
# Selectively interleave large expert/dense slabs across memory controllers.
|
||||
# Unlike blanket numactl interleave, this leaves CUDA staging buffers local.
|
||||
result.setdefault("COLI_NUMA", "1")
|
||||
# NOTE: we intentionally do NOT set OMP_PROC_BIND / OMP_PLACES here.
|
||||
# The engine's own hot-thread tuning (glm.c main(), the COLI_OMP_TUNED
|
||||
# self-exec) sets OMP_PROC_BIND=close with overwrite=0 -- it prefers
|
||||
# packing the team onto adjacent cores for the tiny back-to-back per-expert
|
||||
# matmuls. Pre-setting OMP_PROC_BIND=spread here ran first and won (the
|
||||
# engine's overwrite=0 setenv could not override an already-set var), and
|
||||
# spread + OMP_PLACES=cores collapsed the team to one CPU on some libgomp /
|
||||
# multi-socket topologies (#325: --auto-tier pinned decode to 1 core on a
|
||||
# 64-core box even with OMP_NUM_THREADS=64). Leaving affinity to the engine
|
||||
# makes --auto-tier match the plain (working) path. A user who wants a
|
||||
# specific policy can still set OMP_PROC_BIND/OMP_PLACES in the environment
|
||||
# themselves -- setdefault above only covers OMP_NUM_THREADS.
|
||||
tune = plan.get("tune", {})
|
||||
for key, entry in tune.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
result.setdefault(key, entry["value"])
|
||||
if plan["policy"]["name"] == "balanced":
|
||||
result.setdefault("REPIN", "64")
|
||||
ram = plan["tiers"]["ram"]
|
||||
@@ -386,5 +531,18 @@ def format_plan(plan):
|
||||
else:
|
||||
lines.append("VRAM no NVIDIA device detected · CPU path")
|
||||
lines.append(f"limit {plan['expected_bottleneck']}")
|
||||
hit = plan.get("projected_hit_rate", 0)
|
||||
lines.append(f"hit {hit:.0%} projected expert residency")
|
||||
tune = plan.get("tune", {})
|
||||
if tune:
|
||||
lines.append("")
|
||||
lines.append("auto-tune:")
|
||||
for key, entry in tune.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
lines.append(f" {key}={entry['value']:12s} {entry['reason']}")
|
||||
hint = tune.get("_numa_hint")
|
||||
if hint:
|
||||
lines.append(f" hint: {hint}")
|
||||
lines.extend(f"warn {warning}" for warning in plan["warnings"])
|
||||
return "\n".join(lines)
|
||||
|
||||
+153
@@ -0,0 +1,153 @@
|
||||
/* sample.h — sampling (temperature + nucleus) and stop-set management.
|
||||
* Header-only: all functions are static — include from the main engine file. */
|
||||
#ifndef SAMPLE_H
|
||||
#define SAMPLE_H
|
||||
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include "tok.h"
|
||||
|
||||
/* ---- RNG (xorshift64*) -------------------------------------------------- */
|
||||
static uint64_t g_rng = 0x9E3779B97F4A7C15ULL;
|
||||
static inline double rndu(void){
|
||||
g_rng ^= g_rng << 13; g_rng ^= g_rng >> 7; g_rng ^= g_rng << 17;
|
||||
return (double)(g_rng >> 11) * (1.0 / 9007199254740992.0);
|
||||
}
|
||||
|
||||
/* ---- argmax over a float vector ----------------------------------------- */
|
||||
static inline int argmax_v(const float *lo, int V){
|
||||
int b=-1; float bv=-INFINITY;
|
||||
for(int i=0;i<V;i++){ float x=lo[i]; if(x==x && x>bv){ bv=x; b=i; } }
|
||||
return b<0?0:b;
|
||||
}
|
||||
|
||||
/* ---- distribution buffers (reused, single-threaded decode) --------------- */
|
||||
static float *g_pbuf = NULL;
|
||||
static int *g_pidx = NULL;
|
||||
|
||||
/* sift-down on max-heap in h[0..n), key = g_pbuf[h[i]] (#335: partial top-p).
|
||||
* "hole" variant: carries the root value and deposits only at the end, so
|
||||
* heapify is O(V) and each pop is O(log n) without qsort on the full vocab. */
|
||||
static void topp_siftdown(int *h, int n, int i){
|
||||
int iv = h[i]; float kv = g_pbuf[iv];
|
||||
for (;;) {
|
||||
int l = 2*i + 1;
|
||||
if (l >= n) break;
|
||||
int b = l; if (l+1 < n && g_pbuf[h[l+1]] > g_pbuf[h[l]]) b = l+1;
|
||||
if (g_pbuf[h[b]] <= kv) break;
|
||||
h[i] = h[b]; i = b;
|
||||
}
|
||||
h[i] = iv;
|
||||
}
|
||||
|
||||
/* build the target distribution in g_pbuf: softmax(lo/temp) truncated to
|
||||
* top-p g_nuc. Invariant: g_pbuf stays indexed by token-id (never reordered);
|
||||
* the truncated tail is zeroed (dist_sample reads by id directly).
|
||||
* Requires: g_temp, g_nuc, falloc() — declared in the main engine file. */
|
||||
static void dist_build(const float *lo, int V){
|
||||
if (!g_pbuf) { g_pbuf = falloc(V); g_pidx = malloc(V * sizeof(int)); }
|
||||
int mxi = -1; float mx = 0;
|
||||
for (int i = 0; i < V; i++)
|
||||
if (isfinite(lo[i]) && (mxi < 0 || lo[i] > mx)) { mx = lo[i]; mxi = i; }
|
||||
double s = 0; float invt = 1.f / (g_temp > 1e-4f ? g_temp : 1e-4f);
|
||||
if (mxi >= 0) {
|
||||
for (int i = 0; i < V; i++) {
|
||||
g_pbuf[i] = isfinite(lo[i]) ? expf((lo[i] - mx) * invt) : 0.f;
|
||||
s += g_pbuf[i];
|
||||
}
|
||||
}
|
||||
if (mxi < 0 || !isfinite(s) || s <= 0.0) {
|
||||
static int warned = 0;
|
||||
if (!warned) { warned = 1; fprintf(stderr,
|
||||
"[SAMPLE] warning: non-finite logits (NaN/Inf) — falling back to argmax; "
|
||||
"output may be degraded. This usually means a numerical blow-up upstream.\n"); }
|
||||
int a = (mxi >= 0) ? mxi : 0;
|
||||
for (int i = 0; i < V; i++) g_pbuf[i] = 0.f;
|
||||
g_pbuf[a] = 1.f;
|
||||
return;
|
||||
}
|
||||
for (int i = 0; i < V; i++) g_pbuf[i] /= (float)s;
|
||||
if (g_nuc > 0 && g_nuc < 1.f) {
|
||||
for (int i = 0; i < V; i++) g_pidx[i] = i;
|
||||
for (int i = V/2-1; i >= 0; i--) topp_siftdown(g_pidx, V, i);
|
||||
double s2 = 0, cum = 0; int out = V;
|
||||
do {
|
||||
int root = g_pidx[0];
|
||||
g_pidx[0] = g_pidx[--out]; g_pidx[out] = root;
|
||||
s2 += g_pbuf[root]; cum += g_pbuf[root];
|
||||
if (out > 0) topp_siftdown(g_pidx, out, 0);
|
||||
} while (cum < g_nuc && out > 0);
|
||||
for (int i = 0; i < out; i++) g_pbuf[g_pidx[i]] = 0;
|
||||
float s2f = (float)s2;
|
||||
for (int i = out; i < V; i++) g_pbuf[g_pidx[i]] /= s2f;
|
||||
}
|
||||
}
|
||||
|
||||
/* sample from g_pbuf; ban>=0 excludes that token (renormalizing on the fly) */
|
||||
static int dist_sample(int V, int ban){
|
||||
double z = 1.0 - (ban >= 0 ? g_pbuf[ban] : 0.0);
|
||||
if (z <= 1e-12) z = 1e-12;
|
||||
double u = rndu() * z, cum = 0;
|
||||
for (int i = 0; i < V; i++) { if (i == ban) continue; cum += g_pbuf[i]; if (cum >= u) return i; }
|
||||
for (int i = V-1; i >= 0; i--) if (i != ban && g_pbuf[i] > 0) return i;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* next token from logits: greedy if g_temp<=0, sampling otherwise.
|
||||
* ban = token excluded because it was rejected by speculative verification. */
|
||||
static int pick_tok(const float *lo, int V, int ban){
|
||||
if (g_temp <= 0) return argmax_v(lo, V);
|
||||
dist_build(lo, V);
|
||||
return dist_sample(V, ban);
|
||||
}
|
||||
|
||||
/* ---- stop set ----------------------------------------------------------- */
|
||||
static int g_stop[64], g_nstop = 0;
|
||||
static inline int is_stop(int t){
|
||||
for (int i = 0; i < g_nstop; i++) if (t == g_stop[i]) return 1;
|
||||
return 0;
|
||||
}
|
||||
/* T=NULL -> config stops only (validation/oracle, where the tokenizer is not needed). */
|
||||
static void stops_arm_tok(const Cfg *c, int tok_eos, Tok *T){
|
||||
g_nstop = 0;
|
||||
for (int i = 0; i < c->n_stop && g_nstop < 64; i++) g_stop[g_nstop++] = c->stop_ids[i];
|
||||
if (tok_eos >= 0 && !is_stop(tok_eos) && g_nstop < 64) g_stop[g_nstop++] = tok_eos;
|
||||
int nsp = 0;
|
||||
if (T) for (int id = 0; id < T->n_ids && g_nstop < 64; id++)
|
||||
if (T->id_special[id] && !is_stop(id)) { g_stop[g_nstop++] = id; nsp++; }
|
||||
/* #401: in serve mode keep ONLY <|endoftext|>. Role markers <|user|>/<|observation|>
|
||||
* (config stops + tokenizer special set) are boundaries the Python server owns; as
|
||||
* hard stops they cut generation the moment the model opens a <tool_call> block,
|
||||
* because int4 argmax noise picks a stop-token ID over the correct '<' token. */
|
||||
if (getenv("SERVE") && tok_eos >= 0) {
|
||||
int kept = 0;
|
||||
for (int i = 0; i < g_nstop; i++) if (g_stop[i] == tok_eos) g_stop[kept++] = g_stop[i];
|
||||
if (kept < g_nstop) fprintf(stderr, "[stop] serve mode: filtered %d non-EOS stop tokens (tool-call safety, #401)\n", g_nstop - kept);
|
||||
g_nstop = kept; nsp = 0;
|
||||
}
|
||||
fprintf(stderr, "[stop] %d stop tokens:", g_nstop);
|
||||
for (int i = 0; i < g_nstop; i++) fprintf(stderr, " %d", g_stop[i]);
|
||||
if (nsp) fprintf(stderr, " (%d from the tokenizer's special set)", nsp);
|
||||
fprintf(stderr, "\n");
|
||||
}
|
||||
static void stops_arm(const Cfg *c, int tok_eos){ stops_arm_tok(c, tok_eos, NULL); }
|
||||
|
||||
/* ---- log-prob of a target token given the logit vector ------------------- */
|
||||
static double logprob_target(const float *lo, int V, int target, int *am){
|
||||
float mx = lo[0]; int best = 0;
|
||||
for (int i = 1; i < V; i++) if (lo[i] > mx) { mx = lo[i]; best = i; }
|
||||
double se = 0;
|
||||
for (int i = 0; i < V; i++) se += exp((double)lo[i] - mx);
|
||||
if (am) *am = (best == target);
|
||||
return (double)(lo[target] - mx) - log(se);
|
||||
}
|
||||
|
||||
/* "glm" in model_type, case-insensitive */
|
||||
static int mt_is_glm(const char *s){
|
||||
if (s) for (; *s; s++)
|
||||
if ((s[0]|32) == 'g' && (s[1]|32) == 'l' && (s[2]|32) == 'm') return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif /* SAMPLE_H */
|
||||
+2
-2
@@ -32,11 +32,11 @@ esac
|
||||
|
||||
# 2) build: nativa (veloce, per QUESTA macchina). Per un binario da distribuire: make portable
|
||||
echo " building (ARCH=${ARCH:-native})…"
|
||||
make -s glm ARCH="${ARCH:-native}"
|
||||
make -s colibri ARCH="${ARCH:-native}"
|
||||
|
||||
# 3) self-test sull'oracolo tiny, se presente
|
||||
if [ -d glm_tiny ] && [ -f ref_glm.json ]; then
|
||||
r=$(SNAP=./glm_tiny TF=1 ./glm 64 16 16 2>/dev/null | grep -oE "[0-9]+/[0-9]+ positions" || true)
|
||||
r=$(SNAP=./glm_tiny TF=1 ./colibri 64 16 16 2>/dev/null | grep -oE "[0-9]+/[0-9]+ positions" || true)
|
||||
echo " engine self-test: ${r:-?} (expected 32/32)"
|
||||
fi
|
||||
|
||||
|
||||
@@ -200,7 +200,19 @@ static void st_init(shards *S, const char *snap_dir) {
|
||||
if (a0 < 0 || b0 < a0 || data_start + b0 > fsz) {
|
||||
fprintf(stderr, "%s: tensor '%s' data_offsets [%lld,%lld] out of file bounds (%lld)\n",
|
||||
files[fi], name, (long long)a0, (long long)b0, (long long)fsz); exit(1); }
|
||||
int64_t numel = 1; for (int k = 0; k < shp->len; k++) numel *= (int64_t)shp->kids[k]->num;
|
||||
/* SEC: lo shape viene da un file non fidato (mirror). Senza il guard
|
||||
* di overflow, uno shape tipo [65535,65535,65535,...] fa avvolgere
|
||||
* numel a un valore piccolo/negativo che poi passerebbe il cross-check
|
||||
* numel*esz==nbytes in st_read_f32, riaprendo l'OOB. */
|
||||
int64_t numel = 1; int bad_shape = 0;
|
||||
for (int k = 0; k < shp->len; k++) {
|
||||
int64_t d = (int64_t)shp->kids[k]->num;
|
||||
if (d < 0 || (d != 0 && numel > INT64_MAX / d)) { bad_shape = 1; break; }
|
||||
numel *= d;
|
||||
}
|
||||
if (bad_shape) {
|
||||
fprintf(stderr, "%s: tensor '%s' shape overflows int64 — refusing (hostile or corrupt file)\n",
|
||||
files[fi], name); exit(1); }
|
||||
if (S->n == S->cap) { S->cap *= 2; S->t = realloc(S->t, S->cap*sizeof(st_tensor)); }
|
||||
st_tensor *t = &S->t[S->n++];
|
||||
t->name = strdup(name); t->fd = fd; t->off = data_start + a0;
|
||||
@@ -249,6 +261,15 @@ static void st_prefetch(shards *S, const char *name) {
|
||||
static int64_t st_read_f32(shards *S, const char *name, float *out, int drop) {
|
||||
st_tensor *t = st_find(S, name);
|
||||
if (!t) { fprintf(stderr, "missing tensor: %s\n", name); exit(1); }
|
||||
/* SEC: numel viene dallo shape, nbytes dagli offset — due campi indipendenti
|
||||
* del file. Se non concordano, la memcpy F32 (nbytes) o i loop BF16/F16
|
||||
* (numel elementi da un raw di soli nbytes) sforano il buffer del chiamante,
|
||||
* che e' dimensionato sul config, non sul file. Il chiamante che alloca su
|
||||
* st_numel resta coerente; questo blocca l'ingresso ostile a monte. */
|
||||
int esz = (t->dtype == 2) ? 4 : 2;
|
||||
if (t->numel < 0 || t->numel > t->nbytes / esz || t->numel * (int64_t)esz != t->nbytes) {
|
||||
fprintf(stderr, "%s: tensor '%s' shape/bytes mismatch (numel %lld, %lld bytes, dtype %d) — refusing (hostile or corrupt file)\n",
|
||||
name, name, (long long)t->numel, (long long)t->nbytes, t->dtype); exit(1); }
|
||||
void *raw = malloc(t->nbytes);
|
||||
if (!raw) { fprintf(stderr, "malloc %lld bytes for tensor %s failed\n", (long long)t->nbytes, name); exit(1); }
|
||||
st_pread_full(t->fd, raw, t->nbytes, t->off, "pread data");
|
||||
|
||||
+189
@@ -0,0 +1,189 @@
|
||||
/* telemetry.h — dashboard protocol lines, stats/usage persistence, hardware probe.
|
||||
* Include after Model/Cfg/QT/ESlot/shards and st.h are defined; requires
|
||||
* qt_bytes(), now_s(), rss_gb(), edisk_s(), and the g_cuda_* globals (ifdef). */
|
||||
#ifndef TELEMETRY_H
|
||||
#define TELEMETRY_H
|
||||
|
||||
static int64_t tbytes(int O,int I,int bits){
|
||||
if(bits>=16) return (int64_t)O*I*4;
|
||||
if(bits>=5) return (int64_t)O*I + (int64_t)O*4;
|
||||
return (int64_t)O*((I+1)/2) + (int64_t)O*4;
|
||||
}
|
||||
|
||||
static int64_t expert_bytes_probe(Model *m, int ebits){
|
||||
Cfg *c=&m->c; int64_t eb=0; char nm[256];
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.gate_proj.weight",c->first_dense);
|
||||
if(st_nbytes(&m->S,nm)>0){
|
||||
const char *suf[3]={"gate_proj","up_proj","down_proj"};
|
||||
for(int k=0;k<3;k++){
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.%s.weight",c->first_dense,suf[k]);
|
||||
eb+=st_nbytes(&m->S,nm);
|
||||
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.experts.0.%s.weight.qs",c->first_dense,suf[k]);
|
||||
int64_t q=st_nbytes(&m->S,nm); if(q>0) eb+=q;
|
||||
}
|
||||
}
|
||||
if(eb<=0) eb = tbytes(c->moe_inter,c->hidden,ebits)*2 + tbytes(c->hidden,c->moe_inter,ebits);
|
||||
return eb;
|
||||
}
|
||||
|
||||
/* BRAIN MAP: per-turn expert hit bitmap for the dashboard. */
|
||||
static uint8_t **g_ehit;
|
||||
static void ehit_mark(Model *m, int layer, int eid){
|
||||
if(!g_ehit){ Cfg *c=&m->c;
|
||||
g_ehit=calloc(c->n_layers+1,sizeof(uint8_t*));
|
||||
for(int i=0;i<=c->n_layers;i++) g_ehit[i]=calloc(c->n_experts,1);
|
||||
}
|
||||
g_ehit[layer][eid]=1;
|
||||
}
|
||||
|
||||
/* CPU model + cores + RAM (GB); empty/zero where unavailable. */
|
||||
static void hw_probe(char *cpu, size_t cn, int *cores, double *ram_total, double *ram_avail){
|
||||
cpu[0]=0;
|
||||
#ifdef _WIN32
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
{ unsigned int r[12]={0}; unsigned int *w=r;
|
||||
for(unsigned int f=0x80000002u; f<=0x80000004u; f++,w+=4)
|
||||
__get_cpuid(f,&w[0],&w[1],&w[2],&w[3]);
|
||||
char *b=(char*)r; b[47]=0; while(*b==' ')b++;
|
||||
snprintf(cpu,cn,"%s",b); }
|
||||
#endif
|
||||
#else
|
||||
FILE *ci=fopen("/proc/cpuinfo","r");
|
||||
if(ci){ char ln[256];
|
||||
while(fgets(ln,sizeof(ln),ci)) if(!strncmp(ln,"model name",10)){
|
||||
char *p=strchr(ln,':'); if(p){ p++; while(*p==' ')p++;
|
||||
int n=(int)strlen(p); if(n>0&&p[n-1]=='\n')p[--n]=0;
|
||||
snprintf(cpu,cn,"%s",p); } break; }
|
||||
fclose(ci); }
|
||||
#endif
|
||||
*cores=0;
|
||||
#ifdef _WIN32
|
||||
{ SYSTEM_INFO si; GetSystemInfo(&si); *cores=(int)si.dwNumberOfProcessors; }
|
||||
#elif defined(_SC_NPROCESSORS_ONLN)
|
||||
*cores=(int)sysconf(_SC_NPROCESSORS_ONLN);
|
||||
#endif
|
||||
*ram_total=*ram_avail=0;
|
||||
#ifdef _WIN32
|
||||
compat_meminfo(ram_total,ram_avail);
|
||||
#else
|
||||
FILE *mi=fopen("/proc/meminfo","r");
|
||||
if(mi){ char ln[256]; double mt=0,ma=0;
|
||||
while(fgets(ln,sizeof(ln),mi)){
|
||||
if(sscanf(ln,"MemTotal: %lf",&mt)==1) *ram_total=mt/1e6;
|
||||
if(sscanf(ln,"MemAvailable: %lf",&ma)==1) *ram_avail=ma/1e6;
|
||||
} fclose(mi); }
|
||||
#endif
|
||||
}
|
||||
|
||||
static void hwinfo_emit(Model *m){
|
||||
Cfg *c=&m->c; (void)c;
|
||||
char cpu[256]; int cores; double ram_total,ram_avail;
|
||||
hw_probe(cpu,sizeof(cpu),&cores,&ram_total,&ram_avail);
|
||||
int ngpu=0; double vram_total=0;
|
||||
char gpu_name[128]="";
|
||||
#ifdef COLI_CUDA
|
||||
ngpu=g_cuda_ndev; vram_total=m->gpu_expert_bytes/1e9;
|
||||
for(int i=0;i<g_cuda_ndev;i++){
|
||||
size_t fr=0,to=0; coli_cuda_mem_info(g_cuda_devices[i],&fr,&to);
|
||||
if(!i) vram_total=(double)to*g_cuda_ndev/1e9;
|
||||
}
|
||||
if(g_cuda_ndev>0)
|
||||
snprintf(gpu_name,sizeof(gpu_name),"CUDA device x%d",g_cuda_ndev);
|
||||
#endif
|
||||
printf("HWINFO %d %.1f %.1f %d %.1f %s|%s\n",
|
||||
cores,ram_total,ram_avail,ngpu,vram_total,cpu,gpu_name);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
static void tiers_emit(Model *m){
|
||||
Cfg *c=&m->c; int nsp=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) nsp++;
|
||||
int total=(nsp+(m->has_mtp?1:0))*c->n_experts;
|
||||
int pinned=0,lru=0;
|
||||
for(int i=0;i<=c->n_layers;i++){ pinned+=m->npin?m->npin[i]:0; lru+=m->ecn?m->ecn[i]:0; }
|
||||
int vram=0; double vram_gb=0;
|
||||
#ifdef COLI_CUDA
|
||||
vram=m->gpu_expert_count; vram_gb=m->gpu_expert_bytes/1e9;
|
||||
#endif
|
||||
int ram=pinned-vram+lru; if(ram<0) ram=0;
|
||||
int disk=total-vram-ram; if(disk<0) disk=0;
|
||||
double eb=(double)expert_bytes_probe(m,m->ebits);
|
||||
printf("TIERS %d %d %d %.2f %.2f\n",vram,ram,disk,vram_gb,ram*eb/1e9);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
static void emap_emit(Model *m){
|
||||
Cfg *c=&m->c;
|
||||
int rows=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) rows++;
|
||||
int has_mtp = m->has_mtp && m->eusage[c->n_layers];
|
||||
if(has_mtp) rows++;
|
||||
int cols=c->n_experts;
|
||||
char *hex=malloc((size_t)rows*cols*2+1); int w=0;
|
||||
for(int i=0;i<=c->n_layers;i++){
|
||||
int is_row = (i<c->n_layers && m->L[i].sparse) || (i==c->n_layers && has_mtp);
|
||||
if(!is_row) continue;
|
||||
for(int e=0;e<cols;e++){
|
||||
int tier=0;
|
||||
ESlot *P=m->pin[i];
|
||||
for(int z=0;z<m->npin[i];z++) if(P[z].eid==e){
|
||||
#ifdef COLI_CUDA
|
||||
tier = P[z].g.cuda?2:1;
|
||||
#else
|
||||
tier = 1;
|
||||
#endif
|
||||
break; }
|
||||
if(!tier && m->ecache && m->ecache[i])
|
||||
for(int z=0;z<m->ecn[i];z++) if(m->ecache[i][z].eid==e){ tier=1; break; }
|
||||
uint32_t u = m->eusage[i]?m->eusage[i][e]:0;
|
||||
int heat=0; while(u){ heat++; u>>=1; } if(heat>63) heat=63;
|
||||
int b=(tier<<6)|heat;
|
||||
hex[w++]="0123456789abcdef"[b>>4]; hex[w++]="0123456789abcdef"[b&15];
|
||||
}
|
||||
}
|
||||
hex[w]=0;
|
||||
printf("EMAP %d %d %s\n",rows,cols,hex); fflush(stdout); free(hex);
|
||||
}
|
||||
|
||||
static void hits_emit(Model *m){
|
||||
Cfg *c=&m->c; if(!g_ehit) return;
|
||||
int rows=0;
|
||||
for(int i=0;i<c->n_layers;i++) if(m->L[i].sparse) rows++;
|
||||
int has_mtp = m->has_mtp && m->eusage[c->n_layers];
|
||||
if(has_mtp) rows++;
|
||||
int cols=c->n_experts, nb=(rows*cols+7)/8;
|
||||
uint8_t *bm=calloc(nb,1); int bit=0;
|
||||
for(int i=0;i<=c->n_layers;i++){
|
||||
int is_row = (i<c->n_layers && m->L[i].sparse) || (i==c->n_layers && has_mtp);
|
||||
if(!is_row) continue;
|
||||
for(int e=0;e<cols;e++,bit++)
|
||||
if(g_ehit[i][e]){ bm[bit>>3]|=1<<(bit&7); g_ehit[i][e]=0; }
|
||||
}
|
||||
char *hex=malloc((size_t)nb*2+1); int w=0;
|
||||
for(int b=0;b<nb;b++){ hex[w++]="0123456789abcdef"[bm[b]>>4]; hex[w++]="0123456789abcdef"[bm[b]&15]; }
|
||||
hex[w]=0;
|
||||
printf("HITS %d %d %s\n",rows,cols,hex); fflush(stdout); free(hex); free(bm);
|
||||
}
|
||||
|
||||
static void stats_dump_q(Model *m, const char *path, int quiet){
|
||||
char tmp[2100]; snprintf(tmp,sizeof(tmp),"%s.tmp",path);
|
||||
FILE *f=fopen(tmp,"w"); if(!f){ if(!quiet) perror(tmp); return; }
|
||||
Cfg *c=&m->c; int64_t tot=0, nz=0;
|
||||
for(int i=0;i<=c->n_layers;i++){ if(!m->eusage[i]) continue;
|
||||
for(int e=0;e<c->n_experts;e++) if(m->eusage[i][e]){ fprintf(f,"%d %d %u\n",i,e,m->eusage[i][e]); tot+=m->eusage[i][e]; nz++; } }
|
||||
fclose(f); rename(tmp,path);
|
||||
if(!quiet) fprintf(stderr,"[STATS] %lld selections across %lld distinct experts -> %s\n",(long long)tot,(long long)nz,path);
|
||||
}
|
||||
static void stats_dump(Model *m, const char *path){ stats_dump_q(m,path,0); }
|
||||
|
||||
static char g_usage_path[2100]="";
|
||||
static int64_t usage_load(Model *m, const char *path){
|
||||
FILE *f=fopen(path,"r"); if(!f) return 0;
|
||||
Cfg *c=&m->c; int l,e; uint32_t cnt; int64_t tot=0;
|
||||
while(fscanf(f,"%d %d %u",&l,&e,&cnt)==3)
|
||||
if(l>=0&&l<=c->n_layers&&e>=0&&e<c->n_experts&&m->eusage[l]){ m->eusage[l][e]+=cnt; tot+=cnt; }
|
||||
fclose(f); return tot;
|
||||
}
|
||||
static void usage_save(Model *m){ if(g_usage_path[0]) stats_dump_q(m,g_usage_path,1); }
|
||||
|
||||
#endif /* TELEMETRY_H */
|
||||
@@ -0,0 +1,98 @@
|
||||
# Efficiency suite — regression tests + optimization dossier
|
||||
|
||||
Two layers:
|
||||
|
||||
1. **`test_inefficiency.py`** — tiny-model *asserted* regression tests. Fast
|
||||
(~0.15s/run), gate CI, catch breakage. Run as part of `make test`.
|
||||
2. **`test_efficiency_report.py`** — an *opt-in optimization dossier* for a real
|
||||
model. Runs every instrumentation flag, prints a 9-section report answering
|
||||
*what is doing what, when, with what, is it inefficient, how to improve*.
|
||||
Never fails CI (it's a report, not a gate).
|
||||
|
||||
## The dossier (what you run when optimizing)
|
||||
|
||||
```bash
|
||||
# CPU-only (safe, fast to validate):
|
||||
COLI_EFFICIENCY_MODEL=../glm52_i4_g64 make efficiency-report
|
||||
|
||||
# CUDA (dense + expert tiers — needs a CUDA build, see below):
|
||||
COLI_EFFICIENCY_MODEL=../glm52_i4_g64 COLI_EFFICIENCY_CUDA=1 make efficiency-report
|
||||
```
|
||||
|
||||
It turns ON every observability flag the engine supports — `PROF=1`,
|
||||
`COLI_CUDA_PROFILE=1`, `CACHE_ROUTE=1` (auto-unlocks `route_agree`/`route_kl`),
|
||||
`DISK_SPLIT=1`, `LOOKA=1` — so nothing the engine can tell you is left dark.
|
||||
None of these change the computed output; they only add telemetry.
|
||||
|
||||
The 9 sections, and the question each answers:
|
||||
|
||||
| § | section | answers |
|
||||
|---|---|---|
|
||||
| 1 | PROVENANCE | what is running, on what CPU/backend, with what effective config |
|
||||
| 2 | THROUGHPUT | tok/s + forward-latency p50/p90/p99/max (is the tail healthy?) |
|
||||
| 3 | WHERE TIME GOES | the 5 PROFILE phases as % of decode + absolute seconds + verdict |
|
||||
| 3a | ATTENTION BREAKDOWN | attention split into projection/RoPE, score-softmax-value, output |
|
||||
| 4 | EXPERT CACHE | hit %, experts-loaded/token vs baseline topk |
|
||||
| 5 | DISK I/O | GB fetched, MB/token, GB/s, read-service vs felt-wait, phase split |
|
||||
| 5a | DISK-LOAD SPLIT | loads by decode phase (draft/absorb/verify) + MTP-vs-main bytes |
|
||||
| 6 | ROUTING QUALITY | route_agree %, route_kl, cache swaps |
|
||||
| 6a | ROUTING PREDICTABILITY | LOOKAHEAD recall per predictor (which prefetch wins) |
|
||||
| 7 | SPECULATION | tokens/forward, MTP acceptance % |
|
||||
| 8 | GPU TIERS | resident tensors, expert tier (count/GB/calls), H2D/kernel/D2H ms |
|
||||
|
||||
Every line that crosses an advisory threshold is marked `[FLAG]` with the
|
||||
concrete lever to pull (raise RAM_GB, add PIN_GB, try DIRECT=1, lower CTX, …),
|
||||
and all flags repeat in a summary at the end.
|
||||
|
||||
## Tunable thresholds
|
||||
|
||||
The `IS IT INEFFICIENT?` lines are advisory constants at the top of
|
||||
`test_efficiency_report.py`:
|
||||
|
||||
| constant | default | meaning |
|
||||
|---|---|---|
|
||||
| `DISK_WAIT_DOMINANT` | 0.40 | >40% decode waiting on expert reads → I/O-bound |
|
||||
| `LOW_HIT_RATE` | 0.30 | <30% cache hit → thrashing |
|
||||
| `LOW_ROUTE_AGREE` | 0.80 | <80% routing overlap → prefetch guessing wrong |
|
||||
| `HIGH_TAIL_RATIO` | 3.0 | p99 > 3× p50 → decode stalls |
|
||||
| `LOW_MTP_ACCEPT` | 0.20 | <20% MTP acceptance → draft decoder is dead weight |
|
||||
|
||||
The tiny-model asserted floors live in `tools/efficiency.py` (`TINY_TOK_S_FLOOR`,
|
||||
`MAX_DISK_WAIT_SHARE`, `MIN_CPU_CUDA_AGREEMENT`).
|
||||
|
||||
## The regression tests (what gates CI)
|
||||
|
||||
`test_inefficiency.py` runs on the bundled `glm_tiny` model and asserts:
|
||||
|
||||
- telemetry parses (no format drift)
|
||||
- tiny tok/s ≥ floor (throughput regression)
|
||||
- PROFILE phases present and non-negative (accounting sanity)
|
||||
- disk-wait not dominant on a resident model (I/O-path regression)
|
||||
- CPU determinism (two greedy runs agree)
|
||||
- **CUDA** (skip unless CUDA built): init path, dense uploads VRAM, CPU-vs-CUDA
|
||||
argmax agreement ≥ 70% (kernel-correctness guard)
|
||||
|
||||
```bash
|
||||
make efficiency # tiny CPU tests
|
||||
make efficiency-cuda # tiny CUDA tests (needs CUDA build)
|
||||
```
|
||||
|
||||
## CUDA build prerequisite
|
||||
|
||||
The default `make glm.exe` builds **without** CUDA. The CUDA tests and the CUDA
|
||||
dossier need a host built with `-DCOLI_CUDA` plus the runtime DLL:
|
||||
|
||||
```bash
|
||||
make clean && make glm.exe CUDA_DLL=1 && make cuda-dll
|
||||
```
|
||||
|
||||
`make efficiency-cuda` auto-skips with a clear message if the host is CPU-only
|
||||
(it scans the binary for the "CPU-only" marker the engine embeds).
|
||||
|
||||
## Files
|
||||
|
||||
- `tools/efficiency.py` — shared harness: `parse_run()` (captures every
|
||||
telemetry signal), `run_engine()`, thresholds. Reuses `PROFILE_RE`/`SPEED_RE`
|
||||
from `tools/benchmark_cuda_fixture.py`.
|
||||
- `tests/test_inefficiency.py` — tiny-model asserted tests (CPU + CUDA).
|
||||
- `tests/test_efficiency_report.py` — the opt-in optimization dossier.
|
||||
@@ -24,7 +24,7 @@
|
||||
* Run: make tests/bench_dsa_select && ./tests/bench_dsa_select (not in TEST_BINS)
|
||||
*/
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
* Run: make tests/bench_topp && ./tests/bench_topp (not in TEST_BINS -- not a gate)
|
||||
*/
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -31,7 +31,7 @@
|
||||
* In-memory only (no scratch files), so it builds clean on the Windows MinGW CI
|
||||
* job without the unmerged compat shim. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -0,0 +1,315 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Exhaustive optimization dossier for a colibri engine run.
|
||||
|
||||
This is NOT a pass/fail test. It runs the engine with every instrumentation flag
|
||||
on (PROF, COLI_CUDA_PROFILE, CACHE_ROUTE, DISK_SPLIT, LOOKA) and prints a section-
|
||||
by-section report answering, for each subsystem:
|
||||
|
||||
WHAT is doing it — which phase/kernel/tier
|
||||
WHEN it is doing it — how much of decode wall-time it owns
|
||||
WITH WHAT — the config/weights/tier it used
|
||||
IS IT INEFFICIENT? — a verdict, with the threshold
|
||||
HOW TO IMPROVE — the concrete knob, named
|
||||
|
||||
Activation (opt-in only — NOT in `make test`):
|
||||
COLI_EFFICIENCY_MODEL=<model_dir> python tests/test_efficiency_report.py
|
||||
|
||||
Optional env:
|
||||
COLI_EFFICIENCY_CUDA=1 also exercise the CUDA dense/expert tiers
|
||||
COLI_EFFICIENCY_NGEN=N decode tokens (default 24)
|
||||
COLI_EFFICIENCY_RAM_GB=N RAM budget (default 28)
|
||||
COLI_EFFICIENCY_VRAM_GB=N CUDA expert-tier budget GB (default 4)
|
||||
COLI_EFFICIENCY_PROMPT=... prompt (default: a code-gen prompt)
|
||||
|
||||
Exit code is always 0 (it's a dossier, not a gate). Lines marked FLAG point at
|
||||
the most likely lever to move tok/s for the observed bottleneck.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from tools.efficiency import run_engine, disk_wait_share # noqa: E402
|
||||
|
||||
|
||||
# --- advisory thresholds (the "IS IT INEFFICIENT?" lines) ---
|
||||
DISK_WAIT_DOMINANT = 0.40 # >40% decode waiting on expert reads -> I/O-bound
|
||||
LOW_HIT_RATE = 0.30 # <30% cache hit -> thrashing (cap too small)
|
||||
LOW_ROUTE_AGREE = 0.80 # <80% routing overlap -> prefetch is guessing wrong
|
||||
HIGH_TAIL_RATIO = 3.0 # p99 > 3x p50 -> decode stalls (I/O hiccups / KV grow)
|
||||
LOW_MTP_ACCEPT = 0.20 # <20% MTP acceptance -> draft decoder is dead weight
|
||||
VRAM_WASTE_CALLS = 0 # experts pinned in VRAM but 0 calls served
|
||||
|
||||
|
||||
def _flag(ok): return "OK " if ok else "FLAG"
|
||||
|
||||
|
||||
def _bar(frac, width=24):
|
||||
"""A simple ASCII bar for share visualization."""
|
||||
n = max(0, min(width, round(frac * width)))
|
||||
return "#" * n + "." * (width - n)
|
||||
|
||||
|
||||
def _line(label, value, flag=None, note=""):
|
||||
tag = f" [{flag}]" if flag else ""
|
||||
print(f" {label:<22} {value}{tag} {note}" if note else f" {label:<22} {value}{tag}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
model = os.environ.get("COLI_EFFICIENCY_MODEL")
|
||||
if not model:
|
||||
print(__doc__)
|
||||
print("\nNot activated: set COLI_EFFICIENCY_MODEL=<model_dir> to run.")
|
||||
return 0
|
||||
model = str(Path(model).resolve())
|
||||
if not Path(model).is_dir():
|
||||
print(f"ERROR: {model} is not a directory", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
ngen = int(os.environ.get("COLI_EFFICIENCY_NGEN", "24"))
|
||||
ram_gb = os.environ.get("COLI_EFFICIENCY_RAM_GB", "28")
|
||||
vram_gb = os.environ.get("COLI_EFFICIENCY_VRAM_GB", "4")
|
||||
prompt = os.environ.get(
|
||||
"COLI_EFFICIENCY_PROMPT",
|
||||
"Write a Python function that computes the factorial of a number. "
|
||||
"Include error handling and a docstring.")
|
||||
use_cuda = os.environ.get("COLI_EFFICIENCY_CUDA") == "1"
|
||||
|
||||
# Turn ON every instrumentation flag so the dossier has maximum detail.
|
||||
# These are all observability toggles (PROF/COLI_CUDA_PROFILE/CACHE_ROUTE/
|
||||
# DISK_SPLIT/LOOKA); none change the computed output.
|
||||
overlay = dict(
|
||||
NGEN=str(ngen), TEMP="0", RAM_GB=ram_gb, PROMPT=prompt,
|
||||
PROF="1", CACHE_ROUTE="1", DISK_SPLIT="1", LOOKA="1", ROUTE_AGREE="1",
|
||||
)
|
||||
if use_cuda:
|
||||
overlay.update(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1",
|
||||
COLI_CUDA_PROFILE="1", CUDA_EXPERT_GB=vram_gb)
|
||||
|
||||
print("=" * 78)
|
||||
print(f"OPTIMIZATION DOSSIER — {Path(model).name}")
|
||||
print(f" mode : {'CUDA (dense+expert tiers)' if use_cuda else 'CPU-only'} "
|
||||
f"ngen : {ngen} ram : {ram_gb} GB" +
|
||||
(f" vram : {vram_gb} GB" if use_cuda else ""))
|
||||
print("=" * 78)
|
||||
|
||||
t0 = time.time()
|
||||
t, proc = run_engine(overlay, snap=model, timeout=3600.0)
|
||||
wall = time.time() - t0
|
||||
flags = [] # collected FLAG lines for the summary
|
||||
|
||||
print(f"\n[0] RUN")
|
||||
_line("wall clock", f"{wall:.0f}s")
|
||||
_line("exit code", proc.returncode,
|
||||
None if proc.returncode == 0 else "FLAG",
|
||||
"" if proc.returncode == 0 else "non-zero exit")
|
||||
if proc.returncode != 0:
|
||||
print(" stderr tail:")
|
||||
for ln in proc.stderr.strip().splitlines()[-8:]:
|
||||
print(f" {ln}")
|
||||
return 0
|
||||
|
||||
# ---------------------------------------------------------------- [1] WHO ----
|
||||
print(f"\n[1] PROVENANCE — what is running, on what, with what config")
|
||||
if t.get("machine"):
|
||||
m = t["machine"]
|
||||
_line("CPU", m["cpu"])
|
||||
_line("cores / omp", f"{m['cores']} cores")
|
||||
_line("backend", m["backend"])
|
||||
if t.get("load"):
|
||||
ld = t["load"]
|
||||
_line("model load time", f"{ld['load_s']:.2f}s")
|
||||
_line("resident dense", f"{ld['resident_dense_mb']:.1f} MB")
|
||||
_line("layers / experts", f"{ld['layers']} layers, {ld['experts']} experts")
|
||||
_line("MTP", f"{ld['mtp_status']} (draft={ld['draft']})")
|
||||
if t.get("config_str"):
|
||||
_line("resolved config", t["config_str"])
|
||||
print(" (this is the EFFECTIVE config after auto-budgeting — not your env verbatim)")
|
||||
|
||||
# ---------------------------------------------------------------- [2] SPEED --
|
||||
print(f"\n[2] THROUGHPUT — is it fast, is the tail healthy")
|
||||
if t.get("tok_s") is not None:
|
||||
_line("tok/s", f"{t['tok_s']:.3f}")
|
||||
else:
|
||||
flags.append("throughput line missing — engine output format may have changed")
|
||||
if t.get("latency"):
|
||||
la = t["latency"]
|
||||
_line("decode forwards", f"{int(la['forwards'])}")
|
||||
_line("p50 / p90", f"{la['p50_ms']:.1f} / {la['p90_ms']:.1f} ms")
|
||||
_line("p99 / max", f"{la['p99_ms']:.1f} / {la['max_ms']:.1f} ms")
|
||||
tail_ok = la["p99_ms"] <= HIGH_TAIL_RATIO * la["p50_ms"]
|
||||
_line("tail ratio (p99/p50)", f"{la['p99_ms']/max(la['p50_ms'],1e-9):.2f}x",
|
||||
_flag(tail_ok),
|
||||
"high tail = decode stalls (I/O hiccups, KV growth, re-pin)")
|
||||
if not tail_ok:
|
||||
flags.append(f"tail latency p99={la['p99_ms']:.1f}ms >> p50={la['p50_ms']:.1f}ms "
|
||||
"(look for REPIN swaps or disk stalls)")
|
||||
|
||||
# ---------------------------------------------------------------- [3] TIME ---
|
||||
print(f"\n[3] WHERE TIME GOES — what is doing it, when (share of decode)")
|
||||
ts = t.get("time_shares")
|
||||
prof = t.get("profile")
|
||||
if ts:
|
||||
order = [("io", "expert-disk I/O", DISK_WAIT_DOMINANT, "the cache is too small / disk is slow"),
|
||||
("matmul", "expert matmul", 0.40, "compute-bound; more cores or a GPU expert tier"),
|
||||
("attention", "attention", 0.35, "context length is the cost; lower CTX"),
|
||||
("head", "lm_head", 0.10, "vocab projection; unusual to dominate"),
|
||||
("other", "other", 0.30, "scheduling / KV bookkeeping overhead")]
|
||||
for key, name, thresh, lever in order:
|
||||
f = ts[key]
|
||||
ok = f < thresh
|
||||
_line(name, f"{f:5.0%} {_bar(f)}", _flag(ok),
|
||||
"" if ok else f"->{lever}")
|
||||
if not ok:
|
||||
flags.append(f"{name} dominates ({f:.0%}) -> {lever}")
|
||||
if t.get("verdict"):
|
||||
print(f" engine verdict : {t['verdict']}")
|
||||
elif prof:
|
||||
print(" (no [PROF] time shares — set PROF=1 for phase percentages)")
|
||||
if prof:
|
||||
print(" absolute seconds :")
|
||||
for k in ("disk", "expert_matmul", "attention", "lm_head", "other"):
|
||||
_line(k, f"{prof[k]:.3f}s")
|
||||
|
||||
# attention sub-breakdown: how is attention being read
|
||||
ab = t.get("attn_breakdown")
|
||||
if ab:
|
||||
print(f"\n[3a] ATTENTION BREAKDOWN — how the attention phase is spent")
|
||||
atot = sum(ab.values()) or 1.0
|
||||
for k, label in (("proj_rope", "projection + RoPE"),
|
||||
("score_sm_value", "score-softmax-value"),
|
||||
("out_proj", "output projection")):
|
||||
_line(label, f"{ab[k]:.3f}s ({ab[k]/atot:.0%} of attn)")
|
||||
|
||||
# ---------------------------------------------------------------- [4] CACHE --
|
||||
print(f"\n[4] EXPERT CACHE — is the cache efficient")
|
||||
hit = t.get("hit_pct")
|
||||
if hit is not None:
|
||||
ok = hit >= LOW_HIT_RATE * 100
|
||||
_line("hit rate", f"{hit:.1f}%", _flag(ok),
|
||||
"" if ok else "<30% = thrashing; raise RAM_GB or cap")
|
||||
if not ok:
|
||||
flags.append(f"cache hit {hit:.1f}% is low -> raise RAM_GB (or cap), add PIN_GB")
|
||||
el = t.get("experts_loaded")
|
||||
if el:
|
||||
per_tok = el["per_tok"]
|
||||
_line("experts loaded/token", f"{per_tok:.1f}")
|
||||
_line(" per-layer", f"{el['per_layer']:.2f} across {el['n_sparse_layers']} sparse layers")
|
||||
_line(" baseline", f"topk={el['baseline_topk']} active experts/token")
|
||||
base_topk = el["baseline_topk"]
|
||||
if base_topk > 0 and per_tok > 2 * base_topk:
|
||||
flags.append(f"loading {per_tok:.0f} experts/token vs topk={base_topk} "
|
||||
"-> redundant I/O; cache is re-fetching evicted experts")
|
||||
|
||||
# ---------------------------------------------------------------- [5] DISK ---
|
||||
print(f"\n[5] DISK I/O — is I/O the bottleneck, and where")
|
||||
eio = t.get("expert_io")
|
||||
if eio:
|
||||
_line("total fetched", f"{eio['gb_fetched']:.3f} GB")
|
||||
_line("per token", f"{eio['mb_per_tok']:.1f} MB/token")
|
||||
_line("disk throughput", f"{eio['gb_per_s']:.2f} GB/s over the run")
|
||||
_line("read service", f"{eio['read_service_s']:.2f}s (on I/O threads)")
|
||||
_line("felt wait", f"{eio['felt_wait_s']:.2f}s (stall compute felt)")
|
||||
if eio["felt_wait_s"] > eio["read_service_s"] * 0.5 and eio["read_service_s"] > 0:
|
||||
flags.append("felt wait is a large fraction of read service -> PIPE=1 may not be "
|
||||
"overlapping fully, or DIRECT=1 on NVMe")
|
||||
ds = t.get("disk_split")
|
||||
if ds:
|
||||
print(f"\n[5a] DISK-LOAD SPLIT — which decode phase reads the bytes")
|
||||
_line("draft phase", f"{ds['draft']} loads")
|
||||
_line("absorb phase", f"{ds['absorb']} loads")
|
||||
_line("verify/main", f"{ds['verify_main']} loads")
|
||||
_line("MTP-layer bytes", f"{ds['mtp_loads']} loads, {ds['mtp_gb']:.2f} GB")
|
||||
_line("main-layer bytes", f"{ds['main_loads']} loads, {ds['main_gb']:.2f} GB")
|
||||
if ds.get("mtp_bytes_pct") is not None:
|
||||
_line("MTP share of bytes", f"{ds['mtp_bytes_pct']:.1f}%")
|
||||
share = disk_wait_share(t)
|
||||
if share is not None:
|
||||
ok = share < DISK_WAIT_DOMINANT
|
||||
_line("disk-wait share", f"{share:.0%}", _flag(ok),
|
||||
"" if ok else "I/O-bound (see levers in [3])")
|
||||
|
||||
# ---------------------------------------------------------------- [6] ROUTE --
|
||||
print(f"\n[6] ROUTING QUALITY — is the router / prefetch accurate")
|
||||
ra = t.get("route_agree")
|
||||
if ra:
|
||||
ok = ra["agree_pct"] >= LOW_ROUTE_AGREE * 100
|
||||
_line("route_agree", f"{ra['agree_pct']:.1f}% overlap with true top-K",
|
||||
_flag(ok),
|
||||
"" if ok else "prefetch is guessing wrong; CACHE_ROUTE params may need tuning")
|
||||
_line("route_kl", f"{ra['kl']:.4f} mean KL (lower = closer to true routing)")
|
||||
if not ok:
|
||||
flags.append(f"route_agree {ra['agree_pct']:.1f}% low -> tune ROUTE_J/M/P, "
|
||||
"or prefetch is hurting more than helping")
|
||||
sw = t.get("swap")
|
||||
if sw:
|
||||
_line("cache swaps", f"{sw['swaps']}/{sw['slots']} ({sw['pct']:.1f}%)",
|
||||
None, "high swap = churn between turns")
|
||||
la = t.get("lookahead")
|
||||
if la:
|
||||
print(f"\n[6a] ROUTING PREDICTABILITY — recall of true experts in predicted top-8")
|
||||
print(" (which predictor should drive prefetch? highest recall wins)")
|
||||
for row in la:
|
||||
_line(row["predictor"][:34], f"{row['pct']:5.1f}% ({row['hit']}/{row['tot']})")
|
||||
|
||||
# ---------------------------------------------------------------- [7] SPEC ---
|
||||
print(f"\n[7] SPECULATION — is the draft decoder pulling weight")
|
||||
sp = t.get("speculation")
|
||||
if sp:
|
||||
_line("tokens/forward", f"{sp['tok_per_fw']:.2f} (>1.0 means speculation helps)")
|
||||
_line("forwards/tokens", f"{sp['forwards']} forwards for {sp['tokens']} tokens")
|
||||
# Speculation helps only if acceptance is high enough that tok/forward > 1.
|
||||
# tok_per_fw already == 1.0 when nothing verifies, so judge by acceptance.
|
||||
ok = sp["mtp_accept_pct"] >= LOW_MTP_ACCEPT * 100
|
||||
_line("MTP acceptance", f"{sp['mtp_accept_pct']:.0f}%", _flag(ok),
|
||||
"" if ok else "<20% -> drafts rarely verify; DRAFT=0 may be faster")
|
||||
if not ok:
|
||||
flags.append(f"MTP acceptance {sp['mtp_accept_pct']:.0f}% low -> "
|
||||
"drafts cost more I/O than they save; try DRAFT=0")
|
||||
|
||||
# ---------------------------------------------------------------- [8] GPU ----
|
||||
print(f"\n[8] GPU TIERS — is the GPU actually used")
|
||||
cuda = t.get("cuda") or {}
|
||||
if not cuda.get("enabled"):
|
||||
print(" (CUDA not enabled — CPU-only run)")
|
||||
else:
|
||||
if cuda.get("resident_tensors") is not None:
|
||||
_line("resident dense tensors", f"{cuda['resident_tensors']} tensors, "
|
||||
f"{cuda['resident_gb']:.2f} GB")
|
||||
if cuda.get("expert_count") is not None:
|
||||
waste = cuda["calls_served"] <= VRAM_WASTE_CALLS
|
||||
_line("expert tier", f"{cuda['expert_count']} experts pinned "
|
||||
f"({cuda['expert_gb']:.2f} GB)", _flag(not waste))
|
||||
_line(" calls served", f"{cuda['calls_served']} from VRAM",
|
||||
_flag(not waste),
|
||||
"" if not waste else "WASTE: pinned but never routed -> lower CUDA_EXPERT_GB")
|
||||
if waste:
|
||||
flags.append("VRAM expert tier has 0 calls served -> experts pinned but unused; "
|
||||
"PIN stats may not match this workload")
|
||||
if cuda.get("groups"):
|
||||
g = cuda["groups"]
|
||||
_line("expert groups", f"{g['calls']} calls, {g['experts']} experts, "
|
||||
f"{g['rows']} rows ({g['experts_per_call']:.1f} experts/call)")
|
||||
if cuda.get("groups_timing"):
|
||||
gt = cuda["groups_timing"]
|
||||
_line("GPU timing", f"H2D {gt['h2d_ms']:.1f} ms | kernel {gt['kernel_ms']:.1f} ms | "
|
||||
f"D2H {gt['d2h_ms']:.1f} ms")
|
||||
if gt["h2d_ms"] + gt["d2h_ms"] > gt["kernel_ms"]:
|
||||
flags.append("CUDA H2D+D2H > kernel time -> transfer-bound; "
|
||||
"consider larger expert tier to keep weights resident")
|
||||
|
||||
# ---------------------------------------------------------------- summary ----
|
||||
print("\n" + "=" * 78)
|
||||
if flags:
|
||||
print(f" {len(flags)} FLAG(s) — the most likely levers to move tok/s:")
|
||||
for f in flags:
|
||||
print(f" - {f}")
|
||||
else:
|
||||
print(" no flags — every measured subsystem is within advisory thresholds.")
|
||||
print("=" * 78)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -18,7 +18,7 @@
|
||||
* index, a wrong group boundary or a swapped nibble cannot hide under it —
|
||||
* those are O(1) relative errors, not O(1e-6). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static uint32_t rng_state=0xC0FFEEu;
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@
|
||||
* (sign-trick kernels must treat |−128| as 128 unsigned, not saturate to 127),
|
||||
* and random data at qrow_i8's contract (|x| <= 127, w full int8 range). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static uint32_t rng_state=0x12345678u;
|
||||
|
||||
@@ -0,0 +1,257 @@
|
||||
"""Inefficiency / regression tests for the colibri engine (tiny model, asserted).
|
||||
|
||||
These run against the bundled glm_tiny model (~0.6 MB resident, ~0.1s/run) and
|
||||
gate CI: a regression here means something broke. They run on a plain CPU-only
|
||||
`glm.exe` build; the CUDA_* tests auto-skip if the engine wasn't built with
|
||||
CUDA_DLL=1 (see tests/README_efficiency.md for the build command).
|
||||
|
||||
The signals under test, and the inefficiency each catches:
|
||||
- tok/s floor : a throughput regression (broken build / bad config)
|
||||
- profile phases sum : telemetry accounting bug (other balloons)
|
||||
- disk-wait not dominant: a tiny resident model should never be I/O-bound
|
||||
- CPU determinism : greedy decode is reproducible (no stray RNG/threading)
|
||||
- CUDA init path : COLI_CUDA=1 initializes and does not silently exit 2
|
||||
- CUDA dense uses VRAM : CUDA_DENSE=1 actually uploads tensors (no silent CPU fallback)
|
||||
- CPU vs CUDA TF-match : identical weights+inputs → identical argmax (kernel bug guard)
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from tools.efficiency import (
|
||||
parse_run, run_engine, disk_wait_share, tf_agreement,
|
||||
TINY_TOK_S_FLOOR, MAX_DISK_WAIT_SHARE, MIN_CPU_CUDA_AGREEMENT,
|
||||
)
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
C_DIR = HERE.parent
|
||||
ENGINE = C_DIR / "glm.exe"
|
||||
TINY = C_DIR / "glm_tiny"
|
||||
|
||||
|
||||
def _engine_present() -> bool:
|
||||
"""True iff BOTH the built engine AND the tiny fixture are available.
|
||||
|
||||
These tests need glm.exe (a build artifact) AND glm_tiny/ (a generated
|
||||
fixture, gitignored). CI runs `make check` = "dependency-free tests, no
|
||||
model downloads" (workflow .github/workflows/check.yml, by design #140), so
|
||||
neither is present there and these tests must SKIP rather than fail. They
|
||||
run locally after `make glm.exe` (glm_tiny ships alongside the source, or
|
||||
is regenerated by tools/make_glm_oracle.py).
|
||||
"""
|
||||
return ENGINE.exists() and (TINY / "config.json").exists()
|
||||
|
||||
|
||||
def _skip_reason() -> str:
|
||||
"""Name exactly which prerequisite is missing, so the skip is actionable."""
|
||||
if not ENGINE.exists():
|
||||
return "glm.exe not built (run: make glm.exe)"
|
||||
if not (TINY / "config.json").exists():
|
||||
return "glm_tiny fixture absent (gitignored; ship it locally or run tools/make_glm_oracle.py)"
|
||||
return ""
|
||||
|
||||
|
||||
def _cuda_available() -> bool:
|
||||
"""True iff the engine binary has the CUDA loader compiled in AND the DLL is present.
|
||||
|
||||
The host is built with -DCOLI_CUDA only when CUDA_DLL=1 (Makefile). A binary
|
||||
built without it embeds the string "this binary is CPU-only; rebuild" and
|
||||
exits 2 on any CUDA env var — so we detect the CPU-only build by scanning
|
||||
the binary for that marker (avoids a slow ldd/strings on every import; we
|
||||
only read enough to find it). On Windows the DLL is also required
|
||||
(backend_loader.c loads it at runtime); on Linux it's direct-linked.
|
||||
"""
|
||||
if not _engine_present():
|
||||
return False
|
||||
try:
|
||||
# Read once; the marker is near the read-only string table. 256 KB is
|
||||
# plenty for this string and avoids loading a 1 MB binary into memory.
|
||||
with open(ENGINE, "rb") as f:
|
||||
blob = f.read(2 * 1024 * 1024)
|
||||
if b"this binary is CPU-only" in blob:
|
||||
return False # CPU-only build: COLI_CUDA=1 would exit 2.
|
||||
except OSError:
|
||||
pass
|
||||
# Windows: DLL required at runtime. Linux: direct-linked (no DLL).
|
||||
if os.name == "nt":
|
||||
return (C_DIR / "coli_cuda.dll").exists()
|
||||
import subprocess
|
||||
try:
|
||||
out = subprocess.run(["ldd", str(ENGINE)], capture_output=True, text=True)
|
||||
return "libcudart" in out.stdout
|
||||
except (FileNotFoundError, OSError):
|
||||
return False
|
||||
|
||||
|
||||
@unittest.skipUnless(_engine_present(), _skip_reason() or "glm.exe + glm_tiny required")
|
||||
class TinyEfficiencyTest(unittest.TestCase):
|
||||
"""Asserted regression tests on the resident tiny model. Gates CI."""
|
||||
|
||||
def _run(self, **overlay):
|
||||
return run_engine(overlay, engine=str(ENGINE), snap=str(TINY))[0]
|
||||
|
||||
# -- telemetry contract ---------------------------------------------------
|
||||
|
||||
def test_telemetry_parses(self):
|
||||
"""A REPLAY run must emit the throughput + PROFILE lines the suite keys on.
|
||||
|
||||
If this fails, either the engine changed its output format (update the
|
||||
parsers in tools/efficiency.py) or the run crashed early."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
self.assertIn("tok_s", t["parsed"], f"missing tok/s line:\n{t['stderr']}")
|
||||
self.assertIn("profile", t["parsed"], f"missing PROFILE line:\n{t['stderr']}")
|
||||
self.assertIn("hit_pct", t["parsed"], f"missing expert hit line:\n{t['stderr']}")
|
||||
self.assertIsNotNone(t["tok_s"])
|
||||
self.assertIsNotNone(t["profile"])
|
||||
|
||||
# -- throughput floor -----------------------------------------------------
|
||||
|
||||
def test_tiny_tok_s_floor(self):
|
||||
"""Tiny decode must beat TINY_TOK_S_FLOOR.
|
||||
|
||||
The tiny model is fully resident and runs ~200 tok/s; the 20 tok/s
|
||||
default floor is a 10x margin that catches broken builds or a pathological
|
||||
config cascade (the cap=1 trap from ISSUE_new_model_resource_regression.md)
|
||||
without flapping on machine noise."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="8")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
self.assertGreaterEqual(
|
||||
t["tok_s"], TINY_TOK_S_FLOOR,
|
||||
f"tok/s {t['tok_s']:.1f} below floor {TINY_TOK_S_FLOOR} "
|
||||
f"(regression, or a config cascade starving the cache)",
|
||||
)
|
||||
|
||||
# -- accounting sanity ----------------------------------------------------
|
||||
|
||||
def test_profile_phases_present_and_nonneg(self):
|
||||
"""Every PROFILE phase must be present and non-negative.
|
||||
|
||||
`other` can go slightly negative from timer overhead (the engine allows
|
||||
it), but a large negative means the timers are double-counting."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="4")
|
||||
p = t["profile"]
|
||||
for phase in ("disk", "expert_matmul", "attention", "lm_head"):
|
||||
self.assertGreaterEqual(p[phase], -0.01, f"{phase} went negative: {p}")
|
||||
# 'other' is a residual; allow a small negative from timer overlap.
|
||||
self.assertGreaterEqual(p["other"], -0.05, f"other too negative (double-count): {p}")
|
||||
|
||||
# -- disk-wait not dominant on a resident model ---------------------------
|
||||
|
||||
def test_disk_wait_not_dominant(self):
|
||||
"""A fully-resident tiny model must NOT be I/O-bound.
|
||||
|
||||
Everything fits in RAM; the expert-disk wait share should be ~0. If it
|
||||
exceeds MAX_DISK_WAIT_SHARE, the cache/I/O path regressed — on a real
|
||||
model this same regression would make decode I/O-bound (the exact
|
||||
failure mode the [PROF] verdict flags)."""
|
||||
t = self._run(REPLAY="1", TEMP="0", NGEN="8", PROF="1")
|
||||
self.assertIn("time_shares", t["parsed"], f"missing [PROF] time shares:\n{t['stderr']}")
|
||||
share = disk_wait_share(t)
|
||||
self.assertIsNotNone(share)
|
||||
self.assertLess(
|
||||
share, MAX_DISK_WAIT_SHARE,
|
||||
f"expert-I/O share {share:.0%} on a resident model — I/O path regressed",
|
||||
)
|
||||
|
||||
# -- determinism ----------------------------------------------------------
|
||||
|
||||
def test_cpu_vs_cpu_determinism(self):
|
||||
"""Two greedy REPLAY runs with the same seed produce identical telemetry.
|
||||
|
||||
TEMP=0 = greedy (no sampling), so tok/s and hit-rate must be reproducible.
|
||||
A drift here means non-determinism crept into the decode path (stray
|
||||
threading, uninitialized state) — which on a real model would make A/B
|
||||
comparisons meaningless."""
|
||||
a = self._run(REPLAY="1", TEMP="0", NGEN="8", SEED="1")
|
||||
b = self._run(REPLAY="1", TEMP="0", NGEN="8", SEED="1")
|
||||
self.assertEqual(a["returncode"], 0)
|
||||
self.assertEqual(b["returncode"], 0)
|
||||
self.assertEqual(a["hit_pct"], b["hit_pct"], "greedy hit-rate drifted between runs")
|
||||
# tok/s within 25% — exact equality is too strict across scheduler noise.
|
||||
self.assertLess(abs(a["tok_s"] - b["tok_s"]) / max(a["tok_s"], b["tok_s"]), 0.25)
|
||||
|
||||
|
||||
@unittest.skipUnless(_cuda_available(),
|
||||
_skip_reason() or "CUDA build not present (run: make clean && make glm.exe CUDA_DLL=1 && make cuda-dll)")
|
||||
class TinyCudaEfficiencyTest(unittest.TestCase):
|
||||
"""CUDA-path regression tests on the tiny model. Skip unless CUDA built.
|
||||
|
||||
glm_tiny is small and fully resident, so CUDA here is fast and exercises the
|
||||
real GPU code path (init, dense upload, kernel correctness) without the
|
||||
long load time or memory pressure of the full model. These guard the
|
||||
silent-failure modes that are otherwise invisible:
|
||||
- COLI_CUDA=1 silently falling back to CPU (loader/DLL missing)
|
||||
- CUDA_DENSE=1 uploading nothing
|
||||
- a CUDA kernel producing different argmax than CPU on identical inputs
|
||||
"""
|
||||
|
||||
def _run(self, **overlay):
|
||||
return run_engine(overlay, engine=str(ENGINE), snap=str(TINY))[0]
|
||||
|
||||
def test_cuda_init_path(self):
|
||||
"""COLI_CUDA=1 must initialize the device and NOT exit 2.
|
||||
|
||||
Exit 2 is the engine's "requested backend is unavailable" path
|
||||
(glm.c: g_cuda_enabled check). A clean init prints the [CUDA] device
|
||||
banner to stderr. If this fails, the DLL is broken or the loader can't
|
||||
resolve symbols (ABI drift between backend_cuda.h and the dll)."""
|
||||
t = self._run(COLI_CUDA="1", COLI_GPU="0", REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertNotEqual(t["returncode"], 2,
|
||||
f"engine refused CUDA backend:\n{t['stderr']}")
|
||||
self.assertTrue(t["cuda"]["enabled"],
|
||||
f"no [CUDA] device banner on stderr:\n{t['stderr']}")
|
||||
|
||||
def test_cuda_dense_uses_vram(self):
|
||||
"""CUDA_DENSE=1 must actually upload dense tensors to VRAM.
|
||||
|
||||
The minimal GPU-exercising config (per backend_loader.c analysis):
|
||||
COLI_CUDA=1 + CUDA_DENSE=1. Without CUDA_DENSE the dense path stays on
|
||||
CPU and [CUDA] resident set reports 0 tensors — a silent no-op. This
|
||||
catches that regression: after the run, resident_tensors > 0."""
|
||||
t = self._run(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1",
|
||||
REPLAY="1", TEMP="0", NGEN="4")
|
||||
self.assertEqual(t["returncode"], 0, f"engine exited non-zero:\n{t['stderr']}")
|
||||
rt = t["cuda"]["resident_tensors"]
|
||||
self.assertIsNotNone(rt, f"no [CUDA] resident set line:\n{t['stderr']}")
|
||||
self.assertGreater(rt, 0,
|
||||
f"CUDA_DENSE=1 but {rt} tensors resident — silent CPU fallback")
|
||||
|
||||
def test_cpu_vs_cuda_tf_match(self):
|
||||
"""CPU and CUDA teacher-forcing must agree on most positions (DIRECTLY).
|
||||
|
||||
Both paths prefill the SAME oracle sequence on the SAME weights, so their
|
||||
argmaxes should match position-for-position — but not exactly: the two
|
||||
backends accumulate dot-products in different orders (x86 SIMD vs CUDA
|
||||
kernel), so a few near-tied logits flip. That divergence is expected
|
||||
numeric behavior, not a kernel bug. A *catastrophic* kernel regression
|
||||
(wrong GEMM, wrong scale, wrong fmt) would collapse agreement toward
|
||||
random (~1/vocab = ~4%); the MIN_CPU_CUDA_AGREEMENT floor (default 70%)
|
||||
catches that while tolerating harmless drift.
|
||||
|
||||
We compare CPU-vs-CUDA *directly* (not via the oracle match-counts),
|
||||
because both backends differ from the oracle at different positions and
|
||||
the summary line can't tell "CPU≠CUDA" from "CPU≠oracle"."""
|
||||
import json
|
||||
ref = json.loads((C_DIR / "ref_glm.json").read_text())
|
||||
oracle = ref["tf_pred"]
|
||||
|
||||
cpu = self._run(TF="1", TEMP="0")
|
||||
cuda = self._run(COLI_CUDA="1", COLI_GPU="0", CUDA_DENSE="1", TF="1", TEMP="0")
|
||||
self.assertEqual(cpu["returncode"], 0, f"CPU TF run failed:\n{cpu['stderr']}")
|
||||
self.assertEqual(cuda["returncode"], 0, f"CUDA TF run failed:\n{cuda['stderr']}")
|
||||
self.assertIn("tf_mismatches", cpu["parsed"], "CPU run missing per-position mismatches")
|
||||
self.assertIn("tf_mismatches", cuda["parsed"], "CUDA run missing per-position mismatches")
|
||||
|
||||
agree, diff = tf_agreement(cpu, cuda, oracle)
|
||||
self.assertGreaterEqual(
|
||||
agree, MIN_CPU_CUDA_AGREEMENT,
|
||||
f"CPU-vs-CUDA argmax agreement {agree:.0%} below floor "
|
||||
f"{MIN_CPU_CUDA_AGREEMENT:.0%} — CUDA kernel likely regressed. "
|
||||
f"Differing positions: {diff[:10]}{'...' if len(diff)>10 else ''}",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -4,7 +4,7 @@
|
||||
* arrays twice on the second call -> allocator abort. No model file needed:
|
||||
* the CPU path of kv_alloc only reads c->n_layers/kv_lora/qk_rope. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
int main(void){
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
#include <assert.h>
|
||||
#include <math.h>
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static int approx1(double x){ return x > 0.999 && x < 1.001; }
|
||||
|
||||
@@ -35,7 +35,7 @@ class MakefilePlatformTests(unittest.TestCase):
|
||||
env["PATH"] = ""
|
||||
|
||||
result = subprocess.run(
|
||||
[MAKE, "--no-print-directory", "-B", "-n", "glm"],
|
||||
[MAKE, "--no-print-directory", "-B", "-n", "colibri"],
|
||||
cwd=C_DIR,
|
||||
env=env,
|
||||
text=True,
|
||||
@@ -43,7 +43,7 @@ class MakefilePlatformTests(unittest.TestCase):
|
||||
check=True,
|
||||
)
|
||||
|
||||
self.assertIn("-o glm.exe", result.stdout)
|
||||
self.assertIn("-o colibri.exe", result.stdout)
|
||||
self.assertIn("-fopenmp", result.stdout)
|
||||
self.assertIn("-static", result.stdout)
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
from resource_plan import (
|
||||
GB,
|
||||
@@ -14,6 +15,7 @@ from resource_plan import (
|
||||
environment_for_plan,
|
||||
format_plan,
|
||||
memory_available,
|
||||
physical_cpu_count,
|
||||
)
|
||||
|
||||
|
||||
@@ -87,6 +89,79 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertIn("clamped", plan["warnings"][0])
|
||||
self.assertIn("0:test-gpu", format_plan(plan))
|
||||
|
||||
def test_auto_tier_thread_count_uses_physical_cores(self):
|
||||
# End-to-end for #325: build_plan + environment_for_plan must export the
|
||||
# physical (not logical SMT) core count as OMP_NUM_THREADS. The original
|
||||
# suite passed physical_cpus=24 explicitly, so it never exercised the
|
||||
# real physical_cpu_count() probe whose single-core failure pinned decode.
|
||||
def lscpu(stdout):
|
||||
return subprocess.CompletedProcess(args=[], returncode=0,
|
||||
stdout=stdout, stderr="")
|
||||
# 1 socket, 12 cores, 2 SMT siblings -> 24 threads, 12 physical cores.
|
||||
|
||||
# The parser must return 12 physical cores under BOTH lscpu layouts:
|
||||
# - 2-col: `lscpu -p=core,socket` emits exactly [core,socket] (this is
|
||||
# what the probe actually requests; the previous fields[1]/[2]
|
||||
# indexing skipped every line here and fell through to the
|
||||
# logical count -> the regression JustVugg caught).
|
||||
# - 3-col: bare `lscpu -p` prepends a CPU column -> [cpu,core,socket].
|
||||
# Taking the last two fields is correct in both cases.
|
||||
layouts = {
|
||||
"2-col (-p=core,socket)": (
|
||||
"# core,socket\n" +
|
||||
"\n".join(f"{core},0" for core in range(12) for _ in range(2))),
|
||||
"3-col (bare -p, CPU prefix)": (
|
||||
"# CPU,Core,Socket\n" +
|
||||
"\n".join(f"{cpu},{core},0" for core in range(12) for cpu in range(2))),
|
||||
}
|
||||
for label, blob in layouts.items():
|
||||
with mock.patch("resource_plan.subprocess.run",
|
||||
return_value=lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
plan = build_plan(self.model, available_memory=16 * GB,
|
||||
available_disk=1, gpus=[])
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(plan["cpu"]["physical_cores"], 12, label)
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], "12", label)
|
||||
|
||||
def test_plan_does_not_set_omp_affinity_vars(self):
|
||||
# The real #325 regression: --auto-tier set OMP_PROC_BIND=spread +
|
||||
# OMP_PLACES=cores, which ran before the engine's overwrite=0 setenv and
|
||||
# so won, collapsing the OpenMP team to one CPU on the reporter's 64-core
|
||||
# Linux box even though OMP_NUM_THREADS was correct. The plan must leave
|
||||
# affinity to the engine's own hot-thread tuning (which prefers 'close').
|
||||
plan = build_plan(self.model, available_memory=16 * GB,
|
||||
available_disk=1, gpus=[], physical_cpus=64)
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], "64")
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
|
||||
def test_plan_conserves_budget_and_experts_above_256gb(self):
|
||||
# Regression for #325's reporter: a 512 GB machine loading the whole
|
||||
# model into RAM. Verify the budget math stays exact at large RAM sizes
|
||||
# (no integer truncation, no over-allocation, no experts lost between
|
||||
# tiers). Checked at 256/512/800 GB to bracket the reporter's box.
|
||||
for ram_gb in (256, 512, 800):
|
||||
plan = build_plan(self.model, ram_gb=ram_gb, available_disk=1,
|
||||
gpus=[], physical_cpus=64)
|
||||
ram = plan["tiers"]["ram"]
|
||||
# RAM budget never over-allocated: dense + runtime + cache <= budget.
|
||||
allocated = (ram["dense_bytes"] + ram["runtime_bytes"]
|
||||
+ ram["expert_cache_bytes"])
|
||||
self.assertLessEqual(allocated, ram["budget_bytes"],
|
||||
f"over-allocated RAM at {ram_gb} GB")
|
||||
# Every expert byte is accounted for exactly once across the tiers.
|
||||
tiers = plan["tiers"]
|
||||
tiered = (tiers["vram"]["hot_expert_bytes"]
|
||||
+ ram["warm_expert_bytes"]
|
||||
+ tiers["disk"]["cold_expert_bytes"])
|
||||
self.assertEqual(tiered, plan["model"]["expert_bytes"],
|
||||
f"expert bytes lost/duplicated at {ram_gb} GB")
|
||||
# A positive RAM budget yields a non-negative cache and a sensible cap.
|
||||
self.assertGreaterEqual(ram["expert_cache_bytes"], 0)
|
||||
self.assertGreaterEqual(ram["cache_slots_per_layer"], 0)
|
||||
|
||||
def test_filters_requested_devices(self):
|
||||
gpus = [{"index": 0, "name": "a", "total_bytes": 8 * GB, "free_bytes": 8 * GB}]
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
@@ -117,13 +192,13 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertEqual(env["COLI_CUDA"], "1")
|
||||
self.assertEqual(env["COLI_GPUS"], "1")
|
||||
self.assertEqual(env["OMP_NUM_THREADS"], str(plan["cpu"]["physical_cores"]))
|
||||
if sys.platform == "win32":
|
||||
# MinGW libgomp: niente affinity su Windows, le chiavi non vanno emesse
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
else:
|
||||
self.assertEqual(env["OMP_PROC_BIND"], "spread")
|
||||
self.assertEqual(env["OMP_PLACES"], "cores")
|
||||
# The plan must NOT set OMP_PROC_BIND / OMP_PLACES on any platform:
|
||||
# the engine's own hot-thread tuning owns affinity (it prefers
|
||||
# OMP_PROC_BIND=close for the back-to-back per-expert matmuls). Setting
|
||||
# spread + cores here ran before the engine's overwrite=0 setenv and so
|
||||
# won, collapsing the team to one CPU on some libgomp topologies (#325).
|
||||
self.assertNotIn("OMP_PROC_BIND", env)
|
||||
self.assertNotIn("OMP_PLACES", env)
|
||||
self.assertEqual(env["PIN_GB"], env["CUDA_EXPERT_GB"])
|
||||
|
||||
explicit_threads = environment_for_plan(plan, {"OMP_NUM_THREADS": "7",
|
||||
@@ -140,6 +215,81 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
gpus=[], physical_cpus=8, cpu_sockets=1)
|
||||
self.assertNotIn("COLI_NUMA", environment_for_plan(plan))
|
||||
|
||||
def test_auto_tune_mtp_off_when_compute_bound(self):
|
||||
# Tiny model with 64 GB RAM and no GPU: all experts fit in RAM with no
|
||||
# warm tier, so the plan classifies as compute-bound.
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=24,
|
||||
cpu_sockets=2)
|
||||
# With such a small model fully in RAM and no GPU, bottleneck is compute
|
||||
self.assertEqual(plan["bottleneck_class"], "compute")
|
||||
self.assertIn("DRAFT", plan["tune"])
|
||||
self.assertEqual(plan["tune"]["DRAFT"]["value"], "0")
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["DRAFT"], "0")
|
||||
explicit = environment_for_plan(plan, {"DRAFT": "3"})
|
||||
self.assertEqual(explicit["DRAFT"], "3")
|
||||
|
||||
def test_auto_tune_mtp_off_when_disk_low_hit(self):
|
||||
# Use a model large enough that 8 GB RAM can't hold all experts.
|
||||
big = tempfile.TemporaryDirectory()
|
||||
bigmodel = Path(big.name)
|
||||
(bigmodel / "config.json").write_text(json.dumps({
|
||||
"num_hidden_layers": 2, "n_routed_experts": 4,
|
||||
"kv_lora_rank": 4, "qk_rope_head_dim": 2,
|
||||
"qk_nope_head_dim": 3, "v_head_dim": 5, "num_attention_heads": 2,
|
||||
}))
|
||||
expert_size = 3 * GB # each expert 3 GB → 12 GB total, won't fit in 8 GB budget
|
||||
write_shard(bigmodel / "out-00000.safetensors", [
|
||||
("model.embed_tokens.weight", 100),
|
||||
("model.layers.0.self_attn.q_a_proj.weight", 200),
|
||||
])
|
||||
for i in range(4):
|
||||
write_shard(bigmodel / f"out-{i+1:05d}.safetensors", [
|
||||
(f"model.layers.1.mlp.experts.{i}.gate_proj.weight", expert_size),
|
||||
])
|
||||
plan = build_plan(bigmodel, ram_gb=0, available_memory=4 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=8,
|
||||
cpu_sockets=1)
|
||||
big.cleanup()
|
||||
self.assertEqual(plan["bottleneck_class"], "disk")
|
||||
self.assertLess(plan["projected_hit_rate"], 0.90)
|
||||
self.assertEqual(plan["tune"]["DRAFT"]["value"], "0")
|
||||
|
||||
def test_auto_tune_pipe_multi_gpu(self):
|
||||
gpus = [
|
||||
{"index": 0, "name": "a", "total_bytes": 32 * GB, "free_bytes": 30 * GB},
|
||||
{"index": 1, "name": "b", "total_bytes": 32 * GB, "free_bytes": 30 * GB},
|
||||
]
|
||||
plan = build_plan(self.model, ram_gb=16, available_memory=32 * GB,
|
||||
available_disk=1, gpus=gpus, cpu_sockets=2)
|
||||
self.assertEqual(plan["tune"]["COLI_CUDA_PIPE"]["value"], "2")
|
||||
env = environment_for_plan(plan)
|
||||
self.assertEqual(env["COLI_CUDA_PIPE"], "2")
|
||||
|
||||
def test_auto_tune_pipe_single_gpu(self):
|
||||
gpus = [{"index": 0, "name": "a", "total_bytes": 12 * GB, "free_bytes": 10 * GB}]
|
||||
plan = build_plan(self.model, ram_gb=16, available_memory=32 * GB,
|
||||
available_disk=1, gpus=gpus, cpu_sockets=1)
|
||||
self.assertEqual(plan["tune"]["COLI_CUDA_PIPE"]["value"], "1")
|
||||
|
||||
def test_auto_tune_numa_hint_for_cpu_only(self):
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=1, gpus=[], physical_cpus=64, cpu_sockets=2)
|
||||
self.assertIn("_numa_hint", plan["tune"])
|
||||
self.assertIn("numactl", plan["tune"]["_numa_hint"])
|
||||
self.assertIn("auto-tune", format_plan(plan))
|
||||
|
||||
def test_format_plan_shows_tune_and_hit_rate(self):
|
||||
plan = build_plan(self.model, ram_gb=64, available_memory=64 * GB,
|
||||
available_disk=100 * GB, gpus=[], physical_cpus=24,
|
||||
cpu_sockets=1)
|
||||
text = format_plan(plan)
|
||||
self.assertIn("hit", text)
|
||||
self.assertIn("auto-tune", text)
|
||||
self.assertIn("DRAFT", text)
|
||||
|
||||
def test_cpu_binary_does_not_apply_gpu_tier(self):
|
||||
plan = build_plan(self.model, available_memory=16 * GB, available_disk=1,
|
||||
gpus=[{"index": 0, "name": "a", "total_bytes": 8 * GB,
|
||||
@@ -178,5 +328,73 @@ class ResourcePlanTest(unittest.TestCase):
|
||||
self.assertIn("expected_bottleneck", plan)
|
||||
|
||||
|
||||
class PhysicalCpuCountTest(unittest.TestCase):
|
||||
"""Regression for #325: --auto-tier pinned decode to one core because
|
||||
physical_cpu_count() silently returned 1.
|
||||
|
||||
Two root causes this locks down:
|
||||
1. lscpu -p prepends a CPU column, so `-p=core,socket` emits
|
||||
CPU,Core,Socket; counting rows counted logical SMT siblings.
|
||||
2. any probe failure fell through to ``os.cpu_count() or 1`` and the
|
||||
``or 1`` could pin a constrained/cgroup'd box to a single core.
|
||||
"""
|
||||
|
||||
def _lscpu(self, stdout):
|
||||
return subprocess.CompletedProcess(args=[], returncode=0,
|
||||
stdout=stdout, stderr="")
|
||||
|
||||
def _lscpu_topology(self, sockets, cores_per_socket, threads_per_core):
|
||||
# Real lscpu shape: socket-local core IDs repeat across sockets; the
|
||||
# CPU column (always prepended) is a unique logical-CPU index.
|
||||
rows, cpu = [], 0
|
||||
for sock in range(sockets):
|
||||
for core in range(cores_per_socket):
|
||||
for _ in range(threads_per_core):
|
||||
rows.append(f"{cpu},{core},{sock}")
|
||||
cpu += 1
|
||||
return "# CPU,Core,Socket\n" + "\n".join(rows)
|
||||
|
||||
def test_counts_physical_cores_not_smt_threads(self):
|
||||
blob = self._lscpu_topology(sockets=2, cores_per_socket=16, threads_per_core=2)
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 32)
|
||||
|
||||
def test_single_socket_no_smt(self):
|
||||
blob = self._lscpu_topology(sockets=1, cores_per_socket=8, threads_per_core=1)
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 8)
|
||||
|
||||
def test_skips_offline_core_socket_fields(self):
|
||||
# VMs / large NUMA boxes emit "-" for offline core or socket IDs; that
|
||||
# used to raise ValueError, discard the whole parse, and fall through
|
||||
# to the single-core fallback.
|
||||
blob = "# CPU,Core,Socket\n0,0,0\n1,-,0\n2,1,0\n3,1,0\n"
|
||||
with mock.patch("resource_plan.subprocess.run", return_value=self._lscpu(blob)), \
|
||||
mock.patch.object(sys, "platform", "linux"):
|
||||
self.assertEqual(physical_cpu_count(), 2)
|
||||
|
||||
def test_lscpu_missing_falls_back_to_logical_not_silent_one(self):
|
||||
# The bug: lscpu absent -> os.cpu_count() or 1. On a constrained box
|
||||
# os.cpu_count() can be 1. We still must never silently pick 1 without
|
||||
# a warning, and when logical cores exist they must be used.
|
||||
import os
|
||||
with mock.patch("resource_plan.subprocess.run", side_effect=FileNotFoundError), \
|
||||
mock.patch.object(sys, "platform", "linux"), \
|
||||
mock.patch("resource_plan.os.cpu_count", return_value=16), \
|
||||
mock.patch("sys.stderr"):
|
||||
self.assertEqual(physical_cpu_count(), 16)
|
||||
|
||||
def test_zero_logical_cores_warns_and_returns_one(self):
|
||||
# The genuine degenerate case: no probe works and os.cpu_count() is
|
||||
# None/1. Must return 1 (engine needs a positive team size) but warn.
|
||||
with mock.patch("resource_plan.subprocess.run", side_effect=FileNotFoundError), \
|
||||
mock.patch.object(sys, "platform", "linux"), \
|
||||
mock.patch("resource_plan.os.cpu_count", return_value=None), \
|
||||
mock.patch("sys.stderr"):
|
||||
self.assertEqual(physical_cpu_count(), 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
* iniettato il token scelto e' l'argmax dei FINITI (mai 0 per default), su ogni posizione
|
||||
* del NaN inclusa lo[0]; (c) nessun NaN sopravvive in g_pbuf. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
#include <stdio.h>
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
* Defense 2 is what makes this robust against checkpoints we don't control:
|
||||
* even with BOTH configs mutilated, a control token cannot leak into a reply. */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static const char *TOKJSON =
|
||||
|
||||
+1
-1
@@ -24,7 +24,7 @@
|
||||
* No scratch files: the test runs entirely in memory (no mkdtemp), so it builds clean on
|
||||
* the Windows MinGW CI job without the unmerged compat shim (#352). */
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
#include <math.h>
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
#define main coli_glm_main_unused
|
||||
#include "../glm.c"
|
||||
#include "../colibri.c"
|
||||
#undef main
|
||||
|
||||
static int fail(const char *s){ fprintf(stderr,"FAIL: %s\n",s); return 1; }
|
||||
|
||||
@@ -247,6 +247,32 @@ def convert_shard(path, out_dict, n_layers, ebits, io_bits, xbits,
|
||||
|
||||
def free_gb(p): return shutil.disk_usage(p).free / 1e9
|
||||
|
||||
def check_or_record_params(outdir, prefix, params):
|
||||
"""#383-class guard, mirrored onto the --repo download loops from the --indir
|
||||
path's resume manifest (below): a resumed run with DIFFERENT conversion
|
||||
parameters (bits, group size, PROJ_BITS, ...) must not silently mix bit-widths
|
||||
across shards in the same outdir -- the #355 failure mode (a second pass with
|
||||
changed flags overwriting/interleaving with a finished container in silence).
|
||||
Unlike the --indir manifest this doesn't need to track per-shard completion:
|
||||
the --repo loops already do that via out-NNNNN.safetensors existence, since
|
||||
shard index maps directly to output filename there. Only whether the params
|
||||
used SO FAR match this run's needs checking. Returns False (caller should
|
||||
abort) on a mismatch, True otherwise; records params on first use."""
|
||||
path = os.path.join(outdir, f".{prefix}params.json")
|
||||
if os.path.exists(path):
|
||||
try: prev = json.loads(open(path).read())
|
||||
except (OSError, ValueError): prev = None
|
||||
if prev is not None and prev != params:
|
||||
print(f"ERROR: {path} records a conversion with {prev};\n"
|
||||
f" this run uses {params}. Refusing to mix conversions in the "
|
||||
f"same outdir — use a fresh --outdir (or delete {path} and the "
|
||||
f"{prefix}*.safetensors shards to redo).")
|
||||
return False
|
||||
tmp = path + ".tmp"
|
||||
with open(tmp, "w") as f: json.dump(params, f, indent=1) # atomic write, same reasoning as the --indir manifest
|
||||
os.replace(tmp, path)
|
||||
return True
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--repo", default=None)
|
||||
@@ -440,7 +466,8 @@ def main():
|
||||
# EN: resume skips only what matches, and different parameters on the same
|
||||
# EN: outdir are refused instead of mixing containers (the #355 failure mode).
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map}
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
prog_path = os.path.join(a.outdir, f".{prefix}progress.json")
|
||||
prog = {}
|
||||
if os.path.exists(prog_path):
|
||||
@@ -718,6 +745,10 @@ def main():
|
||||
except Exception: pass
|
||||
tmp = os.path.join(a.outdir, "_inflight"); os.makedirs(tmp, exist_ok=True)
|
||||
if a.mtp:
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-mtp-", params): return
|
||||
import urllib.request
|
||||
idx = json.loads(urllib.request.urlopen(
|
||||
f"https://huggingface.co/{a.repo}/resolve/main/model.safetensors.index.json", timeout=30).read())["weight_map"]
|
||||
@@ -737,6 +768,10 @@ def main():
|
||||
print(f" -> {os.path.basename(outp)} ({os.path.getsize(outp)/1e9:.2f} GB, {len(out)} tensors)", flush=True)
|
||||
shutil.rmtree(tmp, ignore_errors=True); print("[MTP] DONE."); return
|
||||
if a.indexer:
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-idx-", params): return
|
||||
import urllib.request
|
||||
idx = json.loads(urllib.request.urlopen(
|
||||
f"https://huggingface.co/{a.repo}/resolve/main/model.safetensors.index.json", timeout=30).read())["weight_map"]
|
||||
@@ -756,6 +791,10 @@ def main():
|
||||
if os.path.isfile(blob): os.remove(blob)
|
||||
print(f" -> {os.path.basename(outp)} ({len(out)} tensors)", flush=True)
|
||||
shutil.rmtree(tmp, ignore_errors=True); print("[IDX] DONE."); return
|
||||
params = {"ebits": a.ebits, "io_bits": a.io_bits, "xbits": a.xbits,
|
||||
"group_size": a.group_size, "n_layers": a.n_layers, "bits_map": bits_map,
|
||||
"proj_bits": dict(PROJ_BITS)}
|
||||
if not check_or_record_params(a.outdir, "out-", params): return
|
||||
for i, sh in enumerate(shards):
|
||||
if free_gb(a.outdir) < a.min_free_gb:
|
||||
print(f"STOP: free space is below {a.min_free_gb} GB. Free space and rerun to resume."); break
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert OLMoE HuggingFace checkpoint to colibri merged int8 format.
|
||||
|
||||
Consolidates gate_proj, up_proj, and down_proj into a single merged tensor per expert.
|
||||
This allows olmoe.c to load an expert in a single disk read call instead of 3.
|
||||
|
||||
Usage:
|
||||
python tools/convert_olmoe_merged.py --repo allenai/OLMoE-1B-7B-0125-Instruct --out ./olmoe_merged
|
||||
"""
|
||||
|
||||
import argparse, json, os, sys, re
|
||||
from pathlib import Path
|
||||
|
||||
# Windows: force UTF-8 output
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try: s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError): pass
|
||||
|
||||
try:
|
||||
import torch
|
||||
from safetensors.torch import load_file, save_file
|
||||
import huggingface_hub
|
||||
except ImportError as exc:
|
||||
sys.exit(f"Missing dependencies: {exc}. Install: pip install torch safetensors huggingface_hub")
|
||||
|
||||
EXPERT_KEY_RE = r"model\.layers\.(\d+)\.mlp\.experts\.(\d+)\.(gate_proj|up_proj|down_proj)\.weight"
|
||||
|
||||
def quantize_row(w: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
"""Row-wise int8 quantization. Returns (int8_weights, float32_scales)."""
|
||||
w_f32 = w.float()
|
||||
row_max = w_f32.abs().amax(dim=1, keepdim=True).clamp(min=1e-12)
|
||||
scales = row_max / 127.0
|
||||
q = (w_f32 / scales).round().clamp(-128, 127).to(torch.int8)
|
||||
return q, scales.squeeze(1)
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Convert OLMoE HF checkpoint -> colibri merged int8")
|
||||
src = ap.add_mutually_exclusive_group(required=True)
|
||||
src.add_argument("--repo", help="HuggingFace repo ID")
|
||||
src.add_argument("--model", help="Local HF checkpoint directory")
|
||||
ap.add_argument("--out", required=True, help="Output directory for merged model")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.repo:
|
||||
from huggingface_hub import snapshot_download
|
||||
from huggingface_hub.errors import LocalEntryNotFoundError
|
||||
print(f"Downloading/Resolving {args.repo}...")
|
||||
try:
|
||||
src_dir = snapshot_download(args.repo, local_files_only=True, max_workers=4)
|
||||
except LocalEntryNotFoundError:
|
||||
src_dir = None
|
||||
if src_dir is None or not any(Path(src_dir).glob("*.safetensors")):
|
||||
print("Downloading safetensors...")
|
||||
src_dir = snapshot_download(args.repo, max_workers=4)
|
||||
else:
|
||||
src_dir = args.model
|
||||
|
||||
src = Path(src_dir)
|
||||
if not src.is_dir():
|
||||
sys.exit(f"Model directory not found: {src}")
|
||||
if not (src / "config.json").is_file():
|
||||
sys.exit(f"config.json missing in {src}")
|
||||
|
||||
out = Path(args.out)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Copy config.json
|
||||
import shutil
|
||||
shutil.copy2(src / "config.json", out / "config.json")
|
||||
print(f"config.json -> {out}")
|
||||
|
||||
# Process safetensors
|
||||
shards = sorted(src.glob("*.safetensors"))
|
||||
if not shards:
|
||||
sys.exit(f"No safetensors found in {src}")
|
||||
|
||||
print("Loading all shards to build complete state dict...")
|
||||
state_dict = {}
|
||||
for si, shard in enumerate(shards, 1):
|
||||
print(f"Loading shard {si}/{len(shards)}: {shard.name}...")
|
||||
tensors = load_file(str(shard))
|
||||
state_dict.update(tensors)
|
||||
|
||||
# Gather experts
|
||||
experts = {}
|
||||
for name in list(state_dict.keys()):
|
||||
m = re.match(EXPERT_KEY_RE, name)
|
||||
if m:
|
||||
layer_idx, expert_idx, proj = m.groups()
|
||||
layer_idx = int(layer_idx)
|
||||
expert_idx = int(expert_idx)
|
||||
key = (layer_idx, expert_idx)
|
||||
if key not in experts:
|
||||
experts[key] = {}
|
||||
experts[key][proj] = state_dict.pop(name)
|
||||
|
||||
print(f"Found {len(experts)} experts to merge.")
|
||||
|
||||
# Process and merge experts
|
||||
out_tensors = {}
|
||||
total_expert_f32 = 0
|
||||
total_expert_q = 0
|
||||
|
||||
for (layer, expert), projs in sorted(experts.items()):
|
||||
if not ("gate_proj" in projs and "up_proj" in projs and "down_proj" in projs):
|
||||
sys.exit(f"Missing projection for layer {layer} expert {expert}!")
|
||||
|
||||
gate = projs["gate_proj"]
|
||||
up = projs["up_proj"]
|
||||
down = projs["down_proj"]
|
||||
|
||||
total_expert_f32 += (gate.numel() + up.numel() + down.numel()) * gate.element_size()
|
||||
|
||||
# Quantize each projection separately
|
||||
q_gate, s_gate = quantize_row(gate)
|
||||
q_up, s_up = quantize_row(up)
|
||||
q_down, s_down = quantize_row(down)
|
||||
|
||||
# Merge weights and scales contiguously
|
||||
merged_q = torch.cat([q_gate.flatten(), q_up.flatten(), q_down.flatten()])
|
||||
merged_scales = torch.cat([s_gate, s_up, s_down])
|
||||
|
||||
total_expert_q += merged_q.numel() * 1 + merged_scales.numel() * 4
|
||||
|
||||
# Save to output
|
||||
out_tensors[f"model.layers.{layer}.mlp.experts.{expert}.merged_weight"] = merged_q
|
||||
out_tensors[f"model.layers.{layer}.mlp.experts.{expert}.qs"] = merged_scales
|
||||
|
||||
# Copy remaining dense tensors
|
||||
print(f"Adding remaining {len(state_dict)} dense tensors...")
|
||||
out_tensors.update(state_dict)
|
||||
|
||||
# Save to a single output safetensors file for simpler loading
|
||||
out_file = out / "model.safetensors"
|
||||
print(f"Saving merged safetensors model to {out_file}...")
|
||||
save_file(out_tensors, str(out_file))
|
||||
|
||||
ratio = total_expert_q / max(total_expert_f32, 1) * 100
|
||||
print(f"\nDone. {len(experts)} experts successfully merged and saved.")
|
||||
print(f"Expert storage: {total_expert_f32/1e9:.1f} GB -> {total_expert_q/1e9:.1f} GB ({ratio:.0f}%)")
|
||||
print(f"Model ready at: {out}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,458 @@
|
||||
"""Efficiency / regression harness for the colibri engine.
|
||||
|
||||
The engine already emits rich telemetry (REPLAY tok/s, PROFILE phase timings,
|
||||
[PROF] time shares + verdict, CUDA expert-tier utilization). Until now every
|
||||
consumer of that telemetry — `benchmark_cuda_fixture.py`, `bench_full.sh`,
|
||||
`bench_ux.sh` — has only *printed* it for a human to eyeball. This module turns
|
||||
each signal into a parseable field so tests can assert on it.
|
||||
|
||||
Design:
|
||||
- Reuses SPEED_RE / PROFILE_RE from tools.benchmark_cuda_fixture (no drift).
|
||||
- parse_run() is pure: stdout+stderr in, dict out. Easy to unit-test against
|
||||
captured strings (like the existing test_benchmark_cuda_fixture does).
|
||||
- run_engine() is the subprocess wrapper. Captures stdout and stderr
|
||||
separately, because the engine splits them: PROFILE/REPLAY/CUDA-tier go to
|
||||
stdout, the [CUDA]/[PROF]/[prefill] banners go to stderr.
|
||||
- Floor defaults are module constants (tunable in one place, not scattered).
|
||||
|
||||
No model file is required to import this module; only run_engine() invokes the
|
||||
binary. parse_run() works on any captured text, so most test surface is covered
|
||||
by string fixtures without spinning the engine at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
# Reuse the validated regexes from the existing A/B benchmark harness so the
|
||||
# PROFILE field order (disk, expert_matmul, attention, lm_head, other) and the
|
||||
# tok/s capture stay identical. Drift here would silently break every consumer.
|
||||
from tools.benchmark_cuda_fixture import SPEED_RE as _SPEED_RE_REPLAY, PROFILE_RE, PROFILE_KEYS
|
||||
|
||||
# SPEED_RE (from benchmark_cuda_fixture) matches the REPLAY-mode line only:
|
||||
# "REPLAY decode: ... | 12.34 tok/s | ..."
|
||||
# run_text / PROMPT mode uses a DIFFERENT format (glm.c:4682):
|
||||
# "decode N tokens in X.XXs (12.34 tok/s) | expert hit rate ..."
|
||||
# This alt regex catches the parenthesized form so the full-model report (which
|
||||
# uses PROMPT mode) gets a real tok/s instead of reporting it missing.
|
||||
SPEED_RE_TEXT = re.compile(r"decode \d+ tokens in [0-9.]+s \(([0-9.]+) tok/s\)")
|
||||
|
||||
|
||||
def _first_speed(stdout: str):
|
||||
"""Find tok/s in whichever run-mode format the engine used."""
|
||||
for rx in (_SPEED_RE_REPLAY, SPEED_RE_TEXT):
|
||||
m = rx.search(stdout)
|
||||
if m:
|
||||
return m
|
||||
return None
|
||||
|
||||
|
||||
# Public alias so existing imports keep working (tests reference SPEED_RE).
|
||||
SPEED_RE = _SPEED_RE_REPLAY
|
||||
|
||||
|
||||
# --- additional parsers (formats verified against glm.c printf strings) ---
|
||||
|
||||
# "expert hit rate 88.1%" (summary line) | "expert hit 95.0%" (REPLAY line)
|
||||
HIT_RE = re.compile(r"expert hit(?:\s+rate)?\s+([0-9.]+)%")
|
||||
|
||||
# "[PROF] time shares: expert-I/O 3% | expert-matmul 12% | attention 56% | lm_head 2% | other 27%"
|
||||
SHARES_RE = re.compile(
|
||||
r"\[PROF\] time shares: expert-I/O\s+([0-9.]+)%\s*\|\s*expert-matmul\s+([0-9.]+)%\s*"
|
||||
r"\|\s*attention\s+([0-9.]+)%\s*\|\s*lm_head\s+([0-9.]+)%\s*\|\s*other\s+([0-9.]+)%"
|
||||
)
|
||||
|
||||
# "[PROF] verdict: I/O-bound — 60% of the time ..." (also compute-bound / attention-bound / balanced)
|
||||
VERDICT_RE = re.compile(r"\[PROF\] verdict:\s*(I/O-bound|compute-bound|attention-bound|balanced)")
|
||||
|
||||
# "[PROF] expert I/O: ... hit 95.0% (76 hit / 4 load) | 4.0 loads/token"
|
||||
LOADS_PER_TOK_RE = re.compile(r"\|\s*([0-9.]+)\s+loads/token")
|
||||
|
||||
# "CUDA expert tier: 111 resident experts (2.36 GB) | 5400 calls served from VRAM" (stdout)
|
||||
CUDA_TIER_RE = re.compile(
|
||||
r"CUDA expert tier:\s+(\d+)\s+resident experts\s+\(([0-9.]+)\s+GB\)\s*\|\s+(\d+)\s+calls served from VRAM"
|
||||
)
|
||||
|
||||
# "[CUDA] device 0: NVIDIA ..., 14.4 GB VRAM, sm_120" (stderr, per device at init)
|
||||
CUDA_DEVICE_RE = re.compile(r"\[CUDA\] device\s+\d+:")
|
||||
|
||||
# "[CUDA] resident set: 12 tensors, 0.45 GB VRAM" (stderr, cuda_stats_print)
|
||||
CUDA_RESIDENT_RE = re.compile(
|
||||
r"\[CUDA\] resident set:\s+(\d+)\s+tensors,\s+([0-9.]+)\s+GB\s+VRAM"
|
||||
)
|
||||
|
||||
# "PREFILL (teacher-forcing) C vs oracle: 11/32 positions | 1700.4 pos/s" (TF=1 mode, stdout)
|
||||
TF_MATCH_RE = re.compile(r"PREFILL \(teacher-forcing\).*:\s+(\d+)/(\d+)\s+positions")
|
||||
|
||||
# --- the six deeper signals (added after "are you gathering everything?" audit) ---
|
||||
|
||||
# "ATTENTION: projection/RoPE 0.050s | score-softmax-value 0.009s | output projection 0.011s"
|
||||
# Sub-breakdown of the attention phase — answers "how is attention being read".
|
||||
ATTN_BREAKDOWN_RE = re.compile(
|
||||
r"ATTENTION: projection/RoPE\s+([0-9.]+)s\s*\|\s*score-softmax-value\s+([0-9.]+)s\s*"
|
||||
r"\|\s*output projection\s+([0-9.]+)s"
|
||||
)
|
||||
|
||||
# "[PROF] decode forwards: 20 | latency p50 7.3 ms | p90 8.0 ms | p99 8.1 ms | max 8.2 ms | 1.00 tok/forward"
|
||||
# Per-forward tail latency — a p99 >> p50 means decode stalls (I/O hiccups, KV grow).
|
||||
LATENCY_RE = re.compile(
|
||||
r"\[PROF\] decode forwards:\s+(\d+)\s*\| latency p50\s+([0-9.]+)\s*ms\s*\| "
|
||||
r"p90\s+([0-9.]+)\s*ms\s*\|\s*p99\s+([0-9.]+)\s*ms\s*\|\s*max\s+([0-9.]+)\s*ms"
|
||||
)
|
||||
|
||||
# "[PROF] expert I/O: 0.004 GB fetched (0.2 MB/token, 0.03 GB/s over the run) | hit 95.0% ... |
|
||||
# 4.0 loads/token | 0.0s read service / 0.0s felt wait"
|
||||
# Absolute disk throughput + the felt-wait split (the [PROF] version, more detailed than PROFILE).
|
||||
EXPERT_IO_RE = re.compile(
|
||||
r"\[PROF\] expert I/O:\s+([0-9.]+)\s+GB fetched\s+\(([0-9.]+)\s+MB/token,\s*([0-9.]+)\s+GB/s"
|
||||
r".*?\|\s*([0-9.]+)\s+loads/token\s*\|\s*([0-9.]+)s\s+read service\s*/\s*([0-9.]+)s\s+felt wait"
|
||||
)
|
||||
|
||||
# "speculation: 1.05 tokens/forward (19 forwards per 20 tokens) | MTP acceptance 44% (7/16)"
|
||||
# Draft efficiency — is the speculative decoder pulling weight or dead overhead?
|
||||
SPECULATION_RE = re.compile(
|
||||
r"speculation:\s+([0-9.]+)\s+tokens/forward\s+\((\d+)\s+forwards per\s+(\d+)\s+tokens\)"
|
||||
r"\s*\|\s*MTP acceptance\s+([0-9.]+)%"
|
||||
)
|
||||
|
||||
# "experts loaded/token: 450.0 (per-layer 56.25 across 8; baseline topk=8) | TOPK=0 TOPP=0.00"
|
||||
# Fuller than loads_per_tok: includes the per-layer spread + the active topk/topp.
|
||||
EXPERTS_LOADED_RE = re.compile(
|
||||
r"experts loaded/token:\s+([0-9.]+)\s+\(per-layer\s+([0-9.]+)\s+across\s+(\d+);\s*baseline topk=(\d+)\)"
|
||||
)
|
||||
|
||||
# "[PROF] machine: Intel(...) | 22 cores (22 omp threads) | RAM 34.1 GB total, 27.1 GB available | backend CUDA"
|
||||
# Provenance — makes a report reproducible across machines/runs.
|
||||
MACHINE_RE = re.compile(r"\[PROF\] machine:\s*(.*)\|\s*(\d+)\s+cores.*backend\s+(\S+)")
|
||||
|
||||
# "[PROF] config: RAM_GB=auto 23.9 CTX=4096 | expert cache cap 8/layer ... | DRAFT=0 PIPE=1 DIRECT=0 ..."
|
||||
# Effective resolved config (after auto-budgeting). Answers "what config actually ran".
|
||||
CONFIG_RE = re.compile(r"\[PROF\] config:\s*(.*)")
|
||||
|
||||
# "[ORACLE] mismatch pos=7 expected=197 got=22" (TF=1 mode, stderr, per position)
|
||||
# Captures the engine's *actual* argmax at each position, so two backends can be
|
||||
# compared DIRECTLY (independent of how each relates to the oracle).
|
||||
TF_MISMATCH_RE = re.compile(r"\[ORACLE\] mismatch pos=(\d+) expected=\d+ got=(\d+)")
|
||||
|
||||
# --- routing-quality + disk-split + cuda-groups (the deep signals) ---
|
||||
|
||||
# summary suffix: " | swap 3.1% (12/384)" (CACHE_ROUTE inline)
|
||||
SWAP_RE = re.compile(r"swap\s+([0-9.]+)%\s+\((\d+)/(\d+)\)")
|
||||
|
||||
# summary suffix: " | route_agree 85.0% | route_kl 0.0123" (CACHE_ROUTE/ROUTE_AGREE)
|
||||
ROUTE_AGREE_RE = re.compile(r"route_agree\s+([0-9.]+)%\s*\|\s*route_kl\s+([0-9.]+)")
|
||||
|
||||
# "disk-load split: draft 8 + absorb 0 + verify/main 3 misses | MTP-layer 0 loads 0.00 GB |
|
||||
# main-layers 11 loads 0.01 GB (MTP 0.0% of bytes)" (DISK_SPLIT=1)
|
||||
DISK_SPLIT_RE = re.compile(
|
||||
r"disk-load split: draft\s+(\d+)\s+\+\s+absorb\s+(\d+)\s+\+\s+verify/main\s+(\d+)\s+misses"
|
||||
r"\s*\|\s*MTP-layer\s+(\d+)\s+loads\s+([0-9.]+)\s+GB\s*\|\s*main-layers\s+(\d+)\s+loads\s+([0-9.]+)\s+GB"
|
||||
r"(?:\s+\(MTP\s+([0-9.]+)%\s+of bytes\))?"
|
||||
)
|
||||
|
||||
# "[CUDA] expert groups: 120 call, 840 expert, 1200 righe (7.00 expert/call)"
|
||||
CUDA_GROUPS_RE = re.compile(
|
||||
r"\[CUDA\] expert groups:\s+(\d+)\s+call,\s+(\d+)\s+expert,\s+(\d+)\s+righe\s+\(([0-9.]+)\s+expert/call\)"
|
||||
)
|
||||
|
||||
# "[CUDA] expert groups timing: H2D 12.3 ms | kernel 45.6 ms | D2H 7.8 ms" (COLI_CUDA_PROFILE=1)
|
||||
CUDA_GROUPS_TIME_RE = re.compile(
|
||||
r"\[CUDA\] expert groups timing: H2D\s+([0-9.]+)\s+ms\s*\|\s*kernel\s+([0-9.]+)\s+ms\s*\|\s*D2H\s+([0-9.]+)\s+ms"
|
||||
)
|
||||
|
||||
# LOOKAHEAD recall block — 4 named rows. (name, pct, hit, tot)
|
||||
LOOKAHEAD_RE = re.compile(
|
||||
r"^\s*(.+?)\s+([0-9.]+)%\s+\((\d+)/(\d+)\)\s*$", re.MULTILINE
|
||||
)
|
||||
|
||||
# "loaded in 0.02s | resident dense: 0.21 MB | layers=5 experts=8 | MTP absent (draft=0)"
|
||||
LOAD_BANNER_RE = re.compile(
|
||||
r"loaded in\s+([0-9.]+)s\s*\|\s*resident dense:\s+([0-9.]+)\s+MB\s*\|"
|
||||
r"\s*layers=(\d+)\s+experts=(\d+)\s*\|\s*MTP\s+(\w+)\s+\(draft=(\d+)\)"
|
||||
)
|
||||
|
||||
|
||||
# --- tunable floors -----------------------------------------------------------
|
||||
# These are deliberately generous so they catch *regressions* (broken builds,
|
||||
# pathological configs, telemetry accounting bugs) without flapping on machine
|
||||
# noise. The tiny model is fully resident at ~200 tok/s, so a 20 tok/s floor is
|
||||
# a 10x margin. Tune per-host via env if needed (documented in README).
|
||||
TINY_TOK_S_FLOOR = float(os.environ.get("COLI_TINY_TOK_S_FLOOR", "20.0"))
|
||||
# On a fully-resident tiny model the expert-disk wait share should be tiny.
|
||||
# If it exceeds this, something regressed in the I/O accounting or cache path.
|
||||
MAX_DISK_WAIT_SHARE = float(os.environ.get("COLI_MAX_DISK_WAIT_SHARE", "0.20"))
|
||||
# decode wall-time should be roughly the sum of PROFILE phases (other = residual).
|
||||
PROFILE_SUM_TOLERANCE = float(os.environ.get("COLI_PROFILE_SUM_TOL", "0.05"))
|
||||
# Minimum direct CPU-vs-CUDA argmax agreement on the tiny TF fixture. The two
|
||||
# backends use different accumulation orders (SIMD dot vs CUDA kernel), so a
|
||||
# few near-tied positions flip argmax — that's expected numeric divergence, not
|
||||
# a kernel bug. Measured baseline ~84% (27/32) on this fixture; the 70% floor
|
||||
# leaves headroom for machine noise while still catching a catastrophic kernel
|
||||
# regression (e.g. a wrong GEMM would drop this to ~random = ~4%).
|
||||
MIN_CPU_CUDA_AGREEMENT = float(os.environ.get("COLI_MIN_CPU_CUDA_AGREE", "0.70"))
|
||||
|
||||
|
||||
def parse_run(stdout: str, stderr: str = "") -> dict:
|
||||
"""Parse one engine run's output into a telemetry dict.
|
||||
|
||||
Returns keys: tok_s, hit_pct, profile (dict, seconds), profile_sum,
|
||||
time_shares (dict, fractions 0..1), verdict, loads_per_tok, cuda (dict),
|
||||
tf_match (tuple or None), parsed (set of field names found).
|
||||
|
||||
Raises RuntimeError only if the core throughput line is missing — everything
|
||||
else is optional and absent on some run modes (e.g. [PROF] needs PROF=1,
|
||||
CUDA tier needs gpu_expert_count>0).
|
||||
"""
|
||||
out = dict(
|
||||
tok_s=None, hit_pct=None, profile=None, profile_sum=None,
|
||||
time_shares=None, verdict=None, loads_per_tok=None,
|
||||
cuda=None, tf_match=None, stderr=stderr,
|
||||
)
|
||||
parsed = set()
|
||||
blob = stdout + "\n" + stderr # [PROF]/[CUDA] live on stderr; scan both.
|
||||
|
||||
m = _first_speed(stdout)
|
||||
if m:
|
||||
out["tok_s"] = float(m.group(1)); parsed.add("tok_s")
|
||||
|
||||
m = HIT_RE.search(blob)
|
||||
if m:
|
||||
out["hit_pct"] = float(m.group(1)); parsed.add("hit_pct")
|
||||
|
||||
m = PROFILE_RE.search(stdout)
|
||||
if m:
|
||||
service, wait, emm, attn, head, other = (float(x) for x in m.groups())
|
||||
disk = service + (wait or 0.0)
|
||||
out["profile"] = dict(zip(PROFILE_KEYS, (disk, emm, attn, head, other)))
|
||||
out["profile_sum"] = disk + emm + attn + head + other
|
||||
parsed.add("profile")
|
||||
|
||||
# ATTENTION sub-breakdown: projection/RoPE | score-softmax-value | output.
|
||||
m = ATTN_BREAKDOWN_RE.search(stdout)
|
||||
if m:
|
||||
out["attn_breakdown"] = dict(zip(
|
||||
("proj_rope", "score_sm_value", "out_proj"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("attn_breakdown")
|
||||
|
||||
m = SHARES_RE.search(blob)
|
||||
if m:
|
||||
io, emm, attn, head, other = (float(x) / 100.0 for x in m.groups())
|
||||
out["time_shares"] = dict(io=io, matmul=emm, attention=attn, head=head, other=other)
|
||||
parsed.add("time_shares")
|
||||
|
||||
m = VERDICT_RE.search(blob)
|
||||
if m:
|
||||
out["verdict"] = m.group(1); parsed.add("verdict")
|
||||
|
||||
# [PROF] decode forwards + latency p50/p90/p99/max.
|
||||
m = LATENCY_RE.search(blob)
|
||||
if m:
|
||||
out["latency"] = dict(zip(
|
||||
("forwards", "p50_ms", "p90_ms", "p99_ms", "max_ms"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("latency")
|
||||
|
||||
# [PROF] expert I/O throughput: GB fetched, MB/token, GB/s, service vs felt wait.
|
||||
m = EXPERT_IO_RE.search(blob)
|
||||
if m:
|
||||
out["expert_io"] = dict(zip(
|
||||
("gb_fetched", "mb_per_tok", "gb_per_s", "loads_per_tok",
|
||||
"read_service_s", "felt_wait_s"),
|
||||
(float(x) for x in m.groups())))
|
||||
parsed.add("expert_io")
|
||||
|
||||
m = LOADS_PER_TOK_RE.search(blob)
|
||||
if m:
|
||||
out["loads_per_tok"] = float(m.group(1)); parsed.add("loads_per_tok")
|
||||
|
||||
# experts loaded/token with per-layer spread + baseline topk (run_text summary).
|
||||
m = EXPERTS_LOADED_RE.search(stdout)
|
||||
if m:
|
||||
out["experts_loaded"] = dict(
|
||||
per_tok=float(m.group(1)), per_layer=float(m.group(2)),
|
||||
n_sparse_layers=int(m.group(3)), baseline_topk=int(m.group(4)))
|
||||
parsed.add("experts_loaded")
|
||||
|
||||
# speculation: tokens/forward, forwards, tokens, MTP acceptance%.
|
||||
m = SPECULATION_RE.search(stdout)
|
||||
if m:
|
||||
out["speculation"] = dict(zip(
|
||||
("tok_per_fw", "forwards", "tokens", "mtp_accept_pct"),
|
||||
(float(m.group(1)), int(m.group(2)), int(m.group(3)), float(m.group(4)))))
|
||||
parsed.add("speculation")
|
||||
|
||||
# routing quality (CACHE_ROUTE inline suffixes on the summary line).
|
||||
m = SWAP_RE.search(stdout)
|
||||
if m:
|
||||
out["swap"] = dict(pct=float(m.group(1)), swaps=int(m.group(2)), slots=int(m.group(3)))
|
||||
parsed.add("swap")
|
||||
m = ROUTE_AGREE_RE.search(stdout)
|
||||
if m:
|
||||
out["route_agree"] = dict(agree_pct=float(m.group(1)), kl=float(m.group(2)))
|
||||
parsed.add("route_agree")
|
||||
|
||||
# disk-load split by decode phase (DISK_SPLIT=1).
|
||||
m = DISK_SPLIT_RE.search(stdout)
|
||||
if m:
|
||||
out["disk_split"] = dict(zip(
|
||||
("draft", "absorb", "verify_main", "mtp_loads", "mtp_gb",
|
||||
"main_loads", "main_gb", "mtp_bytes_pct"),
|
||||
(int(m.group(1)), int(m.group(2)), int(m.group(3)), int(m.group(4)),
|
||||
float(m.group(5)), int(m.group(6)), float(m.group(7)),
|
||||
float(m.group(8)) if m.group(8) else None)))
|
||||
parsed.add("disk_split")
|
||||
|
||||
# provenance: machine + resolved config.
|
||||
m = MACHINE_RE.search(blob)
|
||||
if m:
|
||||
out["machine"] = dict(cpu=m.group(1).strip(), cores=int(m.group(2)), backend=m.group(3))
|
||||
parsed.add("machine")
|
||||
m = CONFIG_RE.search(blob)
|
||||
if m:
|
||||
out["config_str"] = m.group(1).strip(); parsed.add("config")
|
||||
|
||||
# load banner: load time, resident dense MB, layers, experts, MTP status.
|
||||
m = LOAD_BANNER_RE.search(stdout)
|
||||
if m:
|
||||
out["load"] = dict(zip(
|
||||
("load_s", "resident_dense_mb", "layers", "experts", "mtp_status", "draft"),
|
||||
(float(m.group(1)), float(m.group(2)), int(m.group(3)),
|
||||
int(m.group(4)), m.group(5), int(m.group(6)))))
|
||||
parsed.add("load")
|
||||
|
||||
# LOOKAHEAD routing-recall block (LOOKA=1): list of {predictor, pct, hit, tot}.
|
||||
la_block = re.search(
|
||||
r"LOOKAHEAD routing.*?recall.*?:\n((?:^\s+.+?\s+[0-9.]+%\s+\(\d+/\d+\)\s*$\n?)+)",
|
||||
blob, re.MULTILINE)
|
||||
if la_block:
|
||||
out["lookahead"] = []
|
||||
for row in LOOKAHEAD_RE.finditer(la_block.group(1)):
|
||||
out["lookahead"].append(dict(
|
||||
predictor=row.group(1).strip(), pct=float(row.group(2)),
|
||||
hit=int(row.group(3)), tot=int(row.group(4))))
|
||||
parsed.add("lookahead")
|
||||
|
||||
cuda = dict(enabled=False, expert_count=None, expert_gb=None,
|
||||
calls_served=None, resident_tensors=None, resident_gb=None,
|
||||
groups=None, groups_timing=None)
|
||||
if CUDA_DEVICE_RE.search(stderr):
|
||||
cuda["enabled"] = True
|
||||
m = CUDA_TIER_RE.search(stdout)
|
||||
if m:
|
||||
cuda["expert_count"] = int(m.group(1))
|
||||
cuda["expert_gb"] = float(m.group(2))
|
||||
cuda["calls_served"] = int(m.group(3))
|
||||
m = CUDA_RESIDENT_RE.search(stderr)
|
||||
if m:
|
||||
cuda["resident_tensors"] = int(m.group(1))
|
||||
cuda["resident_gb"] = float(m.group(2))
|
||||
m = CUDA_GROUPS_RE.search(stderr)
|
||||
if m:
|
||||
cuda["groups"] = dict(zip(
|
||||
("calls", "experts", "rows", "experts_per_call"),
|
||||
(int(m.group(1)), int(m.group(2)), int(m.group(3)), float(m.group(4)))))
|
||||
m = CUDA_GROUPS_TIME_RE.search(stderr)
|
||||
if m:
|
||||
cuda["groups_timing"] = dict(zip(
|
||||
("h2d_ms", "kernel_ms", "d2h_ms"),
|
||||
(float(m.group(1)), float(m.group(2)), float(m.group(3)))))
|
||||
out["cuda"] = cuda
|
||||
|
||||
m = TF_MATCH_RE.search(stdout)
|
||||
if m:
|
||||
out["tf_match"] = (int(m.group(1)), int(m.group(2))); parsed.add("tf_match")
|
||||
|
||||
# Capture per-position argmax divergences from the oracle, keyed by position.
|
||||
mismatches = {int(mm.group(1)): int(mm.group(2))
|
||||
for mm in TF_MISMATCH_RE.finditer(blob)}
|
||||
if out["tf_match"] is not None:
|
||||
out["tf_mismatches"] = mismatches
|
||||
parsed.add("tf_mismatches")
|
||||
|
||||
out["parsed"] = parsed
|
||||
return out
|
||||
|
||||
|
||||
def run_engine(
|
||||
env_overlay: dict,
|
||||
*,
|
||||
engine: Optional[str] = None,
|
||||
cap: int = 4,
|
||||
ebits: int = 4,
|
||||
dbits: int = 4,
|
||||
timeout: float = 600.0,
|
||||
snap: Optional[str] = None,
|
||||
) -> tuple[dict, subprocess.CompletedProcess]:
|
||||
"""Run the engine with an env overlay; return (parsed_telemetry, proc).
|
||||
|
||||
`engine` defaults to ./glm.exe (colibri's Windows host). `snap` defaults to
|
||||
the bundled tiny model (glm_tiny) so callers can omit it for fast tests.
|
||||
The positional argv is `cap ebits dbits`, matching the engine's main().
|
||||
"""
|
||||
if engine is None:
|
||||
engine = str(Path(__file__).resolve().parent.parent / "glm.exe")
|
||||
env = os.environ.copy()
|
||||
# Strip CUDA vars by default so a CPU run isn't accidentally GPU-accelerated
|
||||
# by a leftover env; callers opt in by passing them in env_overlay.
|
||||
for k in ("COLI_CUDA", "COLI_GPU", "COLI_GPUS", "CUDA_DENSE", "CUDA_EXPERT_GB"):
|
||||
env.pop(k, None)
|
||||
env.update(env_overlay)
|
||||
if snap is not None:
|
||||
env["SNAP"] = snap
|
||||
elif "SNAP" not in env:
|
||||
env["SNAP"] = str(Path(__file__).resolve().parent.parent / "glm_tiny")
|
||||
|
||||
proc = subprocess.run(
|
||||
[engine, str(cap), str(ebits), str(dbits)],
|
||||
env=env, capture_output=True, text=True, timeout=timeout,
|
||||
)
|
||||
telemetry = parse_run(proc.stdout, proc.stderr)
|
||||
telemetry["returncode"] = proc.returncode
|
||||
telemetry["env"] = {k: env_overlay[k] for k in env_overlay}
|
||||
return telemetry, proc
|
||||
|
||||
|
||||
def disk_wait_share(t: dict) -> Optional[float]:
|
||||
"""Fraction of decode wall-time spent waiting on expert disk reads.
|
||||
|
||||
Preferred source: [PROF] time_shares (the engine's own accounting, which
|
||||
separates felt-wait from read-service). Falls back to PROFILE disk / sum if
|
||||
[PROF] wasn't emitted (PROF=0 runs). None if neither is available.
|
||||
"""
|
||||
if t.get("time_shares"):
|
||||
return t["time_shares"]["io"]
|
||||
if t.get("profile") and t.get("profile_sum"):
|
||||
return t["profile"]["disk"] / t["profile_sum"]
|
||||
return None
|
||||
|
||||
|
||||
def tf_agreement(cpu: dict, cuda: dict, oracle: list[int]) -> tuple[float, list[int]]:
|
||||
"""Direct CPU-vs-CUDA argmax agreement on the TF fixture.
|
||||
|
||||
Both runs prefilled the SAME oracle sequence; tf_mismatches holds each
|
||||
backend's actual argmax where it diverged from the oracle. Where a backend
|
||||
is ABSENT from the mismatch map, its prediction equals the oracle token at
|
||||
that position. So the reconstructed per-position prediction is:
|
||||
oracle[i] if i not in mismatches else mismatches[i]
|
||||
and agreement is the fraction of positions where CPU and CUDA predictions
|
||||
are identical — independent of how each relates to the oracle.
|
||||
|
||||
`oracle` is ref_glm.json's tf_pred (the per-position oracle argmax). Pass
|
||||
n_positions = len(oracle).
|
||||
|
||||
Returns (agreement_fraction, list_of_differing_positions).
|
||||
"""
|
||||
cm = cpu.get("tf_mismatches") or {}
|
||||
gm = cuda.get("tf_mismatches") or {}
|
||||
diff = []
|
||||
for i, orc in enumerate(oracle):
|
||||
cpu_tok = cm.get(i, orc) # matched oracle => oracle token
|
||||
cuda_tok = gm.get(i, orc)
|
||||
if cpu_tok != cuda_tok:
|
||||
diff.append(i)
|
||||
agree = (len(oracle) - len(diff)) / len(oracle) if oracle else 0.0
|
||||
return agree, diff
|
||||
@@ -0,0 +1 @@
|
||||
[[4, 4, 4, 4], [20, 4, 4, 4], [36, 4, 4, 4], [12, 12, 4, 4], [28, 12, 4, 4], [62, 12, 4, 4], [4, 20, 4, 4], [20, 20, 4, 4], [12, 28, 4, 4], [20, 36, 4, 4], [28, 62, 4, 4], [44, 62, 4, 4], [12, 4, 12, 4], [28, 4, 12, 4], [4, 12, 12, 4], [20, 12, 12, 4], [12, 20, 12, 4], [44, 20, 12, 4], [4, 28, 12, 4], [20, 28, 12, 4], [12, 36, 12, 4], [36, 44, 12, 4], [4, 62, 12, 4], [4, 4, 20, 4], [20, 4, 20, 4], [36, 4, 20, 4], [12, 12, 20, 4], [4, 20, 20, 4], [20, 20, 20, 4], [12, 28, 20, 4], [28, 28, 20, 4], [62, 28, 20, 4], [12, 44, 20, 4], [62, 44, 20, 4], [44, 62, 20, 4], [12, 4, 28, 4], [62, 4, 28, 4], [4, 12, 28, 4], [20, 12, 28, 4], [44, 20, 28, 4], [4, 62, 28, 4], [28, 12, 36, 4], [62, 28, 36, 4], [36, 36, 36, 4], [62, 44, 36, 4], [28, 62, 36, 4], [44, 62, 36, 4], [12, 4, 44, 4], [62, 4, 44, 4], [20, 28, 44, 4], [20, 44, 44, 4], [44, 28, 52, 4], [36, 52, 52, 4], [4, 12, 62, 4], [36, 12, 62, 4], [52, 12, 62, 4], [28, 36, 62, 4], [12, 52, 62, 4], [12, 4, 4, 12], [28, 4, 4, 12], [4, 12, 4, 12], [20, 12, 4, 12], [12, 20, 4, 12], [28, 20, 4, 12], [4, 28, 4, 12], [20, 28, 4, 12], [36, 28, 4, 12], [62, 36, 4, 12], [4, 44, 4, 12], [4, 4, 12, 12], [20, 4, 12, 12], [12, 12, 12, 12], [4, 20, 12, 12], [20, 20, 12, 12], [12, 4, 20, 12], [28, 4, 20, 12], [4, 12, 20, 12], [20, 12, 20, 12], [12, 20, 20, 12], [4, 28, 20, 12], [20, 62, 20, 12], [4, 4, 28, 12], [20, 4, 28, 12], [4, 20, 28, 12], [12, 28, 28, 12], [52, 36, 28, 12], [52, 52, 28, 12], [12, 4, 36, 12], [44, 4, 36, 12], [4, 44, 36, 12], [4, 20, 44, 12], [36, 20, 44, 12], [52, 36, 44, 12], [12, 62, 44, 12], [44, 4, 52, 12], [20, 20, 62, 12], [4, 36, 62, 12], [4, 4, 4, 20], [20, 4, 4, 20], [12, 12, 4, 20], [28, 12, 4, 20], [4, 20, 4, 20], [20, 20, 4, 20], [52, 20, 4, 20], [12, 28, 4, 20], [20, 36, 4, 20], [12, 4, 12, 20], [28, 4, 12, 20], [44, 4, 12, 20], [4, 12, 12, 20], [20, 12, 12, 20], [12, 20, 12, 20], [4, 28, 12, 20], [28, 52, 12, 20], [62, 52, 12, 20], [4, 62, 12, 20], [4, 4, 20, 20], [20, 4, 20, 20], [12, 12, 20, 20], [62, 12, 20, 20], [4, 20, 20, 20], [20, 20, 20, 20], [62, 28, 20, 20], [4, 36, 20, 20], [44, 44, 20, 20], [12, 4, 28, 20], [4, 12, 28, 20], [36, 12, 28, 20], [4, 62, 28, 20], [36, 62, 28, 20], [44, 28, 36, 20], [28, 44, 36, 20], [28, 4, 44, 20], [62, 20, 44, 20], [12, 36, 44, 20], [36, 62, 44, 20], [12, 4, 62, 20], [28, 4, 62, 20], [52, 12, 62, 20], [44, 36, 62, 20], [12, 4, 4, 28], [4, 12, 4, 28], [20, 12, 4, 28], [12, 20, 4, 28], [28, 20, 4, 28], [4, 44, 4, 28], [44, 52, 4, 28], [20, 62, 4, 28], [4, 4, 12, 28], [20, 4, 12, 28], [4, 20, 12, 28], [12, 28, 12, 28], [36, 36, 12, 28], [52, 36, 12, 28], [12, 4, 20, 28], [28, 4, 20, 28], [4, 12, 20, 28], [44, 20, 20, 28], [20, 44, 20, 28], [20, 62, 20, 28], [12, 12, 28, 28], [28, 28, 28, 28], [4, 28, 36, 28], [62, 36, 36, 28], [20, 62, 36, 28], [4, 4, 44, 28], [52, 4, 44, 28], [20, 20, 44, 28], [44, 44, 44, 28], [36, 12, 52, 28], [52, 28, 52, 28], [28, 52, 52, 28], [28, 28, 62, 28], [4, 52, 62, 28], [36, 4, 4, 36], [62, 12, 4, 36], [44, 28, 4, 36], [62, 28, 4, 36], [28, 44, 4, 36], [62, 44, 4, 36], [36, 62, 12, 36], [4, 20, 20, 36], [62, 28, 20, 36], [4, 36, 20, 36], [4, 52, 20, 36], [52, 52, 20, 36], [62, 4, 28, 36], [44, 36, 28, 36], [36, 4, 36, 36], [12, 44, 36, 36], [36, 52, 36, 36], [44, 20, 44, 36], [28, 36, 44, 36], [4, 62, 44, 36], [44, 4, 62, 36], [4, 12, 62, 36], [20, 12, 62, 36], [4, 28, 62, 36], [20, 12, 4, 44], [12, 36, 4, 44], [4, 62, 4, 44], [4, 4, 12, 44], [52, 4, 12, 44], [52, 20, 12, 44], [44, 44, 12, 44], [36, 12, 20, 44], [20, 28, 20, 44], [20, 62, 20, 44], [20, 4, 28, 44], [28, 44, 28, 44], [4, 12, 36, 44], [28, 20, 36, 44], [62, 20, 36, 44], [20, 62, 36, 44], [20, 4, 44, 44], [12, 28, 44, 44], [4, 44, 52, 44], [36, 20, 62, 44], [20, 36, 62, 44], [36, 20, 4, 52], [36, 36, 4, 52], [52, 36, 4, 52], [36, 52, 4, 52], [12, 20, 12, 52], [12, 52, 12, 52], [62, 12, 20, 52], [36, 52, 20, 52], [4, 28, 28, 52], [52, 28, 28, 52], [36, 36, 36, 52], [44, 4, 44, 52], [20, 44, 44, 52], [28, 28, 52, 52], [28, 4, 62, 52], [12, 20, 62, 52], [28, 4, 4, 62], [44, 4, 4, 62], [62, 4, 4, 62], [4, 12, 4, 62], [20, 28, 4, 62], [20, 44, 4, 62], [52, 20, 12, 62], [4, 36, 12, 62], [20, 12, 20, 62], [44, 36, 20, 62], [20, 44, 20, 62], [4, 4, 28, 62], [44, 12, 28, 62], [28, 28, 28, 62], [4, 52, 28, 62], [12, 20, 36, 62], [12, 36, 36, 62], [4, 4, 44, 62], [20, 4, 44, 62], [36, 20, 44, 62], [4, 28, 52, 62]]
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Generate reference token IDs for the real OLMoE-1B-7B model.
|
||||
|
||||
Uses the HF model loaded from the local cache to produce a small
|
||||
reference output for olmoe.exe validation. Saves to ref_olmoe_real.json.
|
||||
|
||||
Usage: python tools/make_olmoe_real_oracle.py
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try:
|
||||
s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError):
|
||||
pass
|
||||
|
||||
try:
|
||||
import torch
|
||||
from transformers import AutoTokenizer, OlmoeForCausalLM
|
||||
except ImportError as exc:
|
||||
sys.exit(f"Missing deps: {exc}. Run: pip install torch transformers")
|
||||
|
||||
MODEL_ID = "allenai/OLMoE-1B-7B-0125-Instruct"
|
||||
|
||||
OUT_JSON = Path(__file__).resolve().parent.parent / "ref_olmoe_real.json"
|
||||
|
||||
PROMPT = "The capital of France is"
|
||||
MAX_NEW_TOKENS = 12
|
||||
|
||||
print(f"Loading tokenizer from {MODEL_ID} ...")
|
||||
tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
|
||||
|
||||
print("Encoding prompt ...")
|
||||
enc = tokenizer(PROMPT, return_tensors="pt")
|
||||
prompt_ids = enc["input_ids"][0].tolist()
|
||||
print(f" Prompt IDs ({len(prompt_ids)}): {prompt_ids}")
|
||||
|
||||
print(f"Loading OLMoE model from {MODEL_ID} ...")
|
||||
print(" (this will use ~14 GB RAM — please be patient)")
|
||||
model = OlmoeForCausalLM.from_pretrained(
|
||||
MODEL_ID,
|
||||
torch_dtype=torch.bfloat16,
|
||||
device_map="cpu",
|
||||
low_cpu_mem_usage=True,
|
||||
)
|
||||
model.eval()
|
||||
print(" Model loaded!")
|
||||
|
||||
print(f"Generating {MAX_NEW_TOKENS} tokens ...")
|
||||
with torch.no_grad():
|
||||
out = model.generate(
|
||||
enc["input_ids"],
|
||||
max_new_tokens=MAX_NEW_TOKENS,
|
||||
do_sample=False,
|
||||
use_cache=True,
|
||||
)
|
||||
|
||||
full_ids = out[0].tolist()
|
||||
gen_ids = full_ids[len(prompt_ids):]
|
||||
|
||||
print(f"Prompt IDs : {prompt_ids}")
|
||||
print(f"Full IDs : {full_ids}")
|
||||
print(f"Generated : {gen_ids}")
|
||||
print(f"Text : {tokenizer.decode(gen_ids, skip_special_tokens=True)!r}")
|
||||
|
||||
payload = {"prompt_ids": prompt_ids, "full_ids": full_ids}
|
||||
OUT_JSON.write_text(json.dumps(payload, indent=2))
|
||||
print(f"\nSaved reference to {OUT_JSON}")
|
||||
@@ -117,11 +117,79 @@ def quantize_param(w, bits, group, rot=False, e8=""):
|
||||
|
||||
|
||||
def _grid_or_e8(x, bits, group, e8):
|
||||
if e8 == "-iq3":
|
||||
return _quant_iq3(x.float())
|
||||
if e8:
|
||||
return _quant_e8(x.float(), group, bits, ball=(e8 == "-e8"))
|
||||
return _quant_last_dim(x, bits, group)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------------------
|
||||
# IQ3_XXS-style codebook (#452 candidate (a)): llama.cpp's deployed 3.06-bpw scheme.
|
||||
# 4-dim magnitude blocks quantized to a 256-entry lattice-subset grid (magnitudes on the
|
||||
# odd ladder 4,12,..,62 in half-units), signs factored out per 8 weights with an odd-parity
|
||||
# constraint (7 stored + 1 derived: a block whose true signs violate parity gets its
|
||||
# smallest-magnitude sign flipped — modelled here so the ablation pays the real cost).
|
||||
# Scales: fp16 super-scale per 256 + 4-bit sub-scale per 32, db = d*(0.5+s)*0.5.
|
||||
# Grid extracted from ggml-common.h (MIT).
|
||||
# --------------------------------------------------------------------------------------
|
||||
_IQ3_GRID = None
|
||||
def _iq3_grid(device):
|
||||
global _IQ3_GRID
|
||||
if _IQ3_GRID is None or _IQ3_GRID.device != device:
|
||||
import json, os
|
||||
path = os.path.join(os.path.dirname(__file__), "iq3xxs_grid.json")
|
||||
_IQ3_GRID = torch.tensor(json.load(open(path)), dtype=torch.float32, device=device)
|
||||
return _IQ3_GRID # [256,4], half-unit magnitudes (value/2 = weight units)
|
||||
|
||||
def _quant_iq3(x):
|
||||
orig = x.shape
|
||||
K = orig[-1]
|
||||
assert K % 256 == 0, "iq3 needs multiples of 256 along the input dim"
|
||||
xb = x.reshape(-1, 256) # super-blocks
|
||||
grid = _iq3_grid(x.device) * 0.5 # weight units
|
||||
out = torch.empty_like(xb)
|
||||
signs = torch.sign(xb); signs[signs == 0] = 1.0
|
||||
mags = xb.abs()
|
||||
for sb in range(8): # 8 sub-blocks of 32
|
||||
m = mags[:, sb*32:(sb+1)*32] # [N,32]
|
||||
s = signs[:, sb*32:(sb+1)*32]
|
||||
# per-8 sign parity: flip the smallest-|w| sign where the product is negative
|
||||
s8 = s.reshape(-1, 4, 8)
|
||||
m8 = m.reshape(-1, 4, 8)
|
||||
viol = (s8.prod(-1) < 0) # odd number of minus signs
|
||||
idxmin = m8.argmin(-1)
|
||||
flip = torch.zeros_like(s8)
|
||||
flip.scatter_(-1, idxmin[..., None], 1.0)
|
||||
s8 = torch.where(viol[..., None].expand_as(s8) & (flip > 0), -s8, s8)
|
||||
s = s8.reshape(-1, 32)
|
||||
# sub-scale search: db candidates from the 4-bit code, super d from block RMS
|
||||
d = m.pow(2).mean(-1, keepdim=True).sqrt() / 20.0 + 1e-12 # rough anchor
|
||||
best = None
|
||||
for code in range(16):
|
||||
db = d * (0.5 + code) * 0.5
|
||||
q = m / db # [N,32] target magnitudes
|
||||
q4 = q.reshape(-1, 4) # 4-dim grid blocks
|
||||
# chunked argmin ||q-g||^2 = argmin(|g|^2 - 2 q.g): a full cdist on a
|
||||
# 100M-param tensor materializes tens of GB — this stays at ~256 MB.
|
||||
g2 = grid.pow(2).sum(-1)
|
||||
idx = torch.empty(q4.shape[0], dtype=torch.long, device=q4.device)
|
||||
CH = 1 << 18
|
||||
for i0 in range(0, q4.shape[0], CH):
|
||||
cc = q4[i0:i0+CH]
|
||||
idx[i0:i0+CH] = (g2 - 2.0 * (cc @ grid.T)).argmin(-1)
|
||||
hit = grid[idx].reshape(-1, 8, 4)
|
||||
rec = (hit.reshape(-1, 32) * db)
|
||||
err = (rec - m).pow(2).sum(-1, keepdim=True)
|
||||
if best is None:
|
||||
best = (err, rec)
|
||||
else:
|
||||
take = err < best[0]
|
||||
best = (torch.where(take, err, best[0]), torch.where(take, rec, best[1]))
|
||||
out[:, sb*32:(sb+1)*32] = best[1] * s
|
||||
return out.reshape(orig)
|
||||
|
||||
|
||||
def _rot_quant(x, bits, group, e8=""):
|
||||
"""W -> Qn(W@Q) @ Q^T along the last (input) dim — see rotation() above."""
|
||||
q = rotation(x.shape[-1], x.device)
|
||||
@@ -206,7 +274,7 @@ def _quant_e8(x, group, bits, ball):
|
||||
return best_out.reshape(shp)
|
||||
|
||||
|
||||
SCHEME_RE = re.compile(r"^int(2|3|4|8)(?:-g(\d+))?(-e8u?)?(-rot)?(-nohead)?$")
|
||||
SCHEME_RE = re.compile(r"^int(2|3|4|8)(?:-g(\d+))?(-e8u?|-iq3)?(-rot)?(-nohead)?$")
|
||||
|
||||
|
||||
def parse_scheme(name):
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Bootstrap ref_olmoe_real.json by running olmoe.exe once and capturing output.
|
||||
|
||||
Step 1: Creates a temp ref with only prompt_ids (no full_ids).
|
||||
Step 2: Runs olmoe.exe, parses the generated IDs from stdout.
|
||||
Step 3: Saves {prompt_ids, full_ids} as ref_olmoe_real.json.
|
||||
Step 4: Runs olmoe.exe again against the saved ref to verify determinism.
|
||||
|
||||
No RAM loading of the full model -- the engine streams from SSD as designed.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
if sys.platform == "win32":
|
||||
for s in (sys.stdout, sys.stderr):
|
||||
try:
|
||||
s.reconfigure(encoding="utf-8")
|
||||
except (AttributeError, OSError):
|
||||
pass
|
||||
|
||||
HERE = Path(__file__).resolve().parent.parent
|
||||
ext = ".exe" if sys.platform == "win32" else ""
|
||||
ENGINE = HERE / f"olmoe{ext}"
|
||||
SNAP = os.getenv("SNAP", str(HERE.parent / "olmoe_merged"))
|
||||
REF_OUT = HERE / "ref_olmoe_real.json"
|
||||
BOOTSTRAP_REF = HERE / "ref_olmoe_bootstrap.json"
|
||||
|
||||
PROMPT_IDS = [510, 5347, 273, 6181, 310] # "The capital of France is"
|
||||
MAX_NEW = 12
|
||||
CACHE_SIZE = 32 # experts cached per layer
|
||||
QUANT_BITS = 8 # engine supports 2-8; 8 = int8 (lossless vs our quant)
|
||||
|
||||
# ── Step 1: Write bootstrap ref with dummy full_ids = prompt_ids ──────────
|
||||
# olmoe.exe needs full_ids to know how many tokens to generate (nfull - np).
|
||||
# We extend with MAX_NEW zeros so the engine generates MAX_NEW tokens.
|
||||
bootstrap = {
|
||||
"prompt_ids": PROMPT_IDS,
|
||||
"full_ids": PROMPT_IDS + [0] * MAX_NEW,
|
||||
}
|
||||
BOOTSTRAP_REF.write_text(json.dumps(bootstrap))
|
||||
print(f"Bootstrap ref written to {BOOTSTRAP_REF}")
|
||||
|
||||
env = {**os.environ, "SNAP": str(SNAP)}
|
||||
|
||||
# ── Step 2: Run engine once to capture generated IDs ─────────────────────
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Run 1/2 — capturing engine output (cache={CACHE_SIZE}, bits={QUANT_BITS}) ...")
|
||||
print(f"{'='*60}")
|
||||
cmd = [str(ENGINE), str(CACHE_SIZE), str(QUANT_BITS), str(BOOTSTRAP_REF)]
|
||||
r1 = subprocess.run(cmd, env=env, capture_output=True, text=True, cwd=str(HERE))
|
||||
print(r1.stdout)
|
||||
if r1.returncode != 0:
|
||||
print("STDERR:", r1.stderr, file=sys.stderr)
|
||||
sys.exit(r1.returncode)
|
||||
|
||||
# Parse "C engine : <id> <id> ..." line
|
||||
m = re.search(r"C engine\s*:\s*([\d ]+)", r1.stdout)
|
||||
if not m:
|
||||
sys.exit("Could not parse 'C engine :' line from output")
|
||||
gen_ids = [int(x) for x in m.group(1).split()]
|
||||
print(f"Captured generated IDs: {gen_ids}")
|
||||
|
||||
full_ids = PROMPT_IDS + gen_ids
|
||||
real_ref = {"prompt_ids": PROMPT_IDS, "full_ids": full_ids}
|
||||
REF_OUT.write_text(json.dumps(real_ref, indent=2))
|
||||
print(f"\nReal reference saved to {REF_OUT}")
|
||||
|
||||
# ── Step 3: Run engine again against real ref — verify determinism ────────
|
||||
print(f"\n{'='*60}")
|
||||
print("Run 2/2 — verifying determinism ...")
|
||||
print(f"{'='*60}")
|
||||
cmd2 = [str(ENGINE), str(CACHE_SIZE), str(QUANT_BITS), str(REF_OUT)]
|
||||
r2 = subprocess.run(cmd2, env=env, capture_output=True, text=True, cwd=str(HERE))
|
||||
print(r2.stdout)
|
||||
if r2.returncode != 0:
|
||||
print("STDERR:", r2.stderr, file=sys.stderr)
|
||||
sys.exit(r2.returncode)
|
||||
|
||||
if "Matching tokens: 12/12" in r2.stdout or f"Matching tokens: {MAX_NEW}/{MAX_NEW}" in r2.stdout:
|
||||
print("✓ Engine is DETERMINISTIC — same output on both runs!")
|
||||
else:
|
||||
m2 = re.search(r"Matching tokens: (\d+)/(\d+)", r2.stdout)
|
||||
if m2:
|
||||
print(f"⚠ Partial match: {m2.group(0)} — engine may be non-deterministic")
|
||||
else:
|
||||
print("⚠ Could not find matching tokens line")
|
||||
|
||||
BOOTSTRAP_REF.unlink(missing_ok=True)
|
||||
@@ -0,0 +1,5 @@
|
||||
"""colibrì — tiny engine, immense model."""
|
||||
|
||||
from colibri._version import __version__
|
||||
|
||||
__all__ = ["__version__"]
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Version accessor for the pip package.
|
||||
|
||||
The single source of truth is c/version.py (#394): coli --version and the
|
||||
GitHub Release workflow read it, so the pip metadata must read the SAME file
|
||||
instead of carrying a second literal that would drift on the first bump.
|
||||
|
||||
From a checkout (the supported install: `pip install -e .`) the file is read
|
||||
directly. From an installed wheel c/ is not on disk, so fall back to the
|
||||
package metadata that setuptools baked at build time from that same file.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
_ns = {}
|
||||
exec((Path(__file__).resolve().parent.parent / "c" / "version.py").read_text(), _ns)
|
||||
__version__ = _ns["__version__"]
|
||||
except OSError:
|
||||
from importlib.metadata import PackageNotFoundError, version
|
||||
|
||||
try:
|
||||
__version__ = version("colibri-engine")
|
||||
except PackageNotFoundError:
|
||||
__version__ = "0.0.0+unknown"
|
||||
@@ -0,0 +1,30 @@
|
||||
"""Entry point for `coli` when installed via pip.
|
||||
|
||||
Delegates to the original c/coli script which handles all subcommands.
|
||||
This wrapper exists so `pip install colibri-engine` creates a `coli` console
|
||||
script that works without the user having to add c/ to PATH manually.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import runpy
|
||||
|
||||
|
||||
def main():
|
||||
here = os.path.dirname(os.path.abspath(__file__))
|
||||
engine_dir = os.path.join(os.path.dirname(here), "c")
|
||||
coli_script = os.path.join(engine_dir, "coli")
|
||||
|
||||
if not os.path.exists(coli_script):
|
||||
sys.exit(
|
||||
"colibri engine directory not found.\n"
|
||||
"Install from source: git clone + pip install -e ."
|
||||
)
|
||||
|
||||
sys.path.insert(0, engine_dir)
|
||||
sys.argv[0] = coli_script
|
||||
runpy.run_path(coli_script, run_name="__main__")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,184 @@
|
||||
# Quick Start — from zero to a running model
|
||||
|
||||
A step-by-step guide for first-time users on **Linux**, **Windows**, and **macOS**.
|
||||
No prior experience with C, CUDA, or model conversion is assumed. If you get
|
||||
stuck, `./coli doctor` (below) tells you exactly what's missing.
|
||||
|
||||
> **What you're setting up:** colibrì runs a very large Mixture-of-Experts model
|
||||
> (e.g. GLM-5.2, 744B parameters) on a normal machine by streaming the model's
|
||||
> experts from disk instead of needing them all in RAM. The engine is a single
|
||||
> C program; Python is only used once, to prepare the model files.
|
||||
|
||||
---
|
||||
|
||||
## 0. What you need first (prerequisites)
|
||||
|
||||
| | Minimum | Recommended |
|
||||
|---|---|---|
|
||||
| **RAM** | ~16 GB | 24 GB+ |
|
||||
| **Free disk** | ~380 GB for the int4 model | a fast NVMe SSD (streaming speed = your token speed) |
|
||||
| **OS** | Linux, Windows 10/11, or macOS | any |
|
||||
| **Tools** | a C compiler + `make` + `git` + `python3` | — |
|
||||
|
||||
You do **not** need a GPU. A GPU only helps if you have one; the engine runs
|
||||
CPU-only by default.
|
||||
|
||||
---
|
||||
|
||||
## 1. Install the build tools
|
||||
|
||||
### Linux (Ubuntu / Debian)
|
||||
|
||||
```bash
|
||||
sudo apt update
|
||||
sudo apt install -y build-essential git python3
|
||||
```
|
||||
|
||||
`build-essential` gives you `gcc`, `make`, and OpenMP (libgomp) — everything the
|
||||
engine needs.
|
||||
|
||||
### Windows
|
||||
|
||||
You have two options.
|
||||
|
||||
**Option A — download a prebuilt binary (no compiler needed).**
|
||||
Grab `colibri-<version>-windows-x86_64.zip` from the
|
||||
[Releases page](https://github.com/JustVugg/colibri/releases) and unzip it.
|
||||
Inside you'll find:
|
||||
|
||||
| File | What it is |
|
||||
|---|---|
|
||||
| `colibri-<version>-windows-x86_64.exe` | **the engine** — the C program that actually runs the model |
|
||||
| `coli` | the command-line launcher (`chat`, `serve`, `convert`, `doctor`, …) |
|
||||
| `openai_server.py`, `resource_plan.py`, `doctor.py` | Python support for the API server and placement planner |
|
||||
|
||||
Two setup steps:
|
||||
|
||||
1. **Rename the engine to `glm.exe`** so the launcher can find it (it looks for a
|
||||
binary named `glm`):
|
||||
```powershell
|
||||
Rename-Item colibri-*-windows-x86_64.exe glm.exe
|
||||
```
|
||||
2. **Install Python 3** from [python.org](https://www.python.org/downloads/) — the
|
||||
`coli` launcher and the API gateway are Python scripts (the engine itself is
|
||||
pure C and needs nothing).
|
||||
|
||||
Then continue to [step 3](#3-get-the-model). Prefer to skip the launcher? You can
|
||||
run the engine directly — `.\glm.exe` reads the model path from the `SNAP`
|
||||
environment variable (see [docs/windows.md](windows.md)) — but `coli chat` is the
|
||||
easy path.
|
||||
|
||||
**Option B — build from source with MSYS2.**
|
||||
Install [MSYS2](https://www.msys2.org/), open the **UCRT64** shell, and run:
|
||||
|
||||
```bash
|
||||
pacman -S --needed mingw-w64-ucrt-x86_64-gcc make git python
|
||||
```
|
||||
|
||||
### macOS
|
||||
|
||||
```bash
|
||||
xcode-select --install # C compiler (clang)
|
||||
brew install libomp git python # OpenMP for multithreading
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 2. Get the code and build the engine
|
||||
|
||||
```bash
|
||||
git clone https://github.com/JustVugg/colibri.git
|
||||
cd colibri/c
|
||||
./setup.sh
|
||||
```
|
||||
|
||||
`setup.sh` checks your compiler and OpenMP, builds the engine, and runs a tiny
|
||||
self-test. When it prints:
|
||||
|
||||
```
|
||||
engine self-test: 32/32 (expected 32/32)
|
||||
```
|
||||
|
||||
the engine is working correctly. (On Windows Option A you already have the
|
||||
binary — you can skip this step.)
|
||||
|
||||
---
|
||||
|
||||
## 3. Get the model
|
||||
|
||||
You have two paths.
|
||||
|
||||
### Easiest — download a ready-made int4 container
|
||||
|
||||
A pre-converted **GLM-5.2 int4** model is on Hugging Face. **Use the version
|
||||
with the int8 MTP heads** (the plain int4 heads disable speculative decoding —
|
||||
see [#8](https://github.com/JustVugg/colibri/issues/8)):
|
||||
|
||||
**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp**
|
||||
|
||||
Download it into a folder on a fast disk, e.g. `/nvme/glm52_i4` (Linux/macOS) or
|
||||
`D:\glm52_i4` (Windows). It is about **372 GB**, so make sure you have the space.
|
||||
|
||||
### Or convert it yourself from the FP8 source
|
||||
|
||||
One resumable command downloads and converts the model shard by shard, so it
|
||||
never needs the full ~756 GB on disk at once:
|
||||
|
||||
```bash
|
||||
./coli convert --model /nvme/glm52_i4
|
||||
```
|
||||
|
||||
This step uses Python and runs only once. Safe to interrupt and re-run — it
|
||||
resumes where it left off.
|
||||
|
||||
---
|
||||
|
||||
## 4. Run it
|
||||
|
||||
Point `COLI_MODEL` at the folder from step 3 and start chatting:
|
||||
|
||||
```bash
|
||||
# Linux / macOS
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat
|
||||
|
||||
# Windows (UCRT64 shell)
|
||||
COLI_MODEL=/d/glm52_i4 ./coli chat
|
||||
```
|
||||
|
||||
Useful first commands:
|
||||
|
||||
```bash
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli doctor # read-only check: is everything ready?
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli plan # shows where the model will live (RAM/disk/GPU)
|
||||
COLI_MODEL=/nvme/glm52_i4 ./coli chat --topp 0.85 # faster: reads less from disk, same quality
|
||||
```
|
||||
|
||||
> **Tip:** `--topp 0.85` is worth adding on a disk-bound machine — it reads
|
||||
> fewer expert bytes per token with no quality loss, which directly means more
|
||||
> tokens per second.
|
||||
|
||||
---
|
||||
|
||||
## 5. What to expect
|
||||
|
||||
- **First launch loads the resident weights** (~10 GB) — this takes a moment.
|
||||
- **Speed depends on your disk.** The experts stream from storage, so a fast
|
||||
NVMe SSD is the single biggest factor in tokens/second. On a slow or shared
|
||||
disk, generation can be well under 1 token/second — that's expected, and it's
|
||||
the honest cost of running a 744B model on a small machine.
|
||||
- **It's still the full model.** Placement only changes speed, never the model's
|
||||
answers or precision.
|
||||
|
||||
If something doesn't work, run `./coli doctor` — it reports exactly what's
|
||||
missing (compiler, model files, permissions) and how to fix it.
|
||||
|
||||
---
|
||||
|
||||
## Where to go next
|
||||
|
||||
| Topic | Doc |
|
||||
|---|---|
|
||||
| Windows native build (and CUDA DLL) | [docs/windows.md](windows.md) |
|
||||
| Tuning: cache, prefetch, speculation | [docs/tuning.md](tuning.md) |
|
||||
| OpenAI-compatible API + web dashboard | [docs/api.md](api.md) |
|
||||
| Every environment variable | [docs/ENVIRONMENT.md](ENVIRONMENT.md) |
|
||||
@@ -0,0 +1,53 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=68.0"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "colibri-engine"
|
||||
dynamic = ["version"]
|
||||
description = "Tiny engine, immense model — run GLM-5.2 (744B MoE) locally"
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
requires-python = ">=3.10"
|
||||
authors = [
|
||||
{name = "JustVugg"},
|
||||
]
|
||||
classifiers = [
|
||||
"Development Status :: 4 - Beta",
|
||||
"Environment :: Console",
|
||||
"Intended Audience :: Science/Research",
|
||||
"Operating System :: POSIX :: Linux",
|
||||
"Operating System :: MacOS",
|
||||
"Operating System :: Microsoft :: Windows",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
convert = [
|
||||
"numpy",
|
||||
"huggingface_hub",
|
||||
]
|
||||
oracle = [
|
||||
"torch>=2.0",
|
||||
"transformers>=4.40",
|
||||
"safetensors",
|
||||
]
|
||||
bench = [
|
||||
"tokenizers",
|
||||
"datasets",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
coli = "colibri.cli:main"
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/JustVugg/colibri"
|
||||
Issues = "https://github.com/JustVugg/colibri/issues"
|
||||
|
||||
[tool.setuptools.dynamic]
|
||||
version = {attr = "colibri._version.__version__"}
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["."]
|
||||
include = ["colibri*"]
|
||||
+64
-57
@@ -9,6 +9,7 @@ import {
|
||||
Database,
|
||||
Feather,
|
||||
Gauge,
|
||||
Globe,
|
||||
HardDrive,
|
||||
KeyRound,
|
||||
Layers,
|
||||
@@ -34,6 +35,7 @@ import { Brain } from "./Brain"
|
||||
import { Profiling } from "./Profiling"
|
||||
import { persistPublicSettings, stored } from "@/lib/storage"
|
||||
import { cn } from "@/lib/utils"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
const message = (role: ChatMessage["role"], content: string): ChatMessage => {
|
||||
let id: string
|
||||
@@ -42,15 +44,12 @@ const message = (role: ChatMessage["role"], content: string): ChatMessage => {
|
||||
}
|
||||
|
||||
export default function App() {
|
||||
// When the page is served by the engine itself (coli web), same-origin is the
|
||||
// right default: no CORS, no manual endpoint editing. The Vite dev server
|
||||
// (port 5173) keeps the classic default.
|
||||
const { t, locale, setLocale, locales } = useLocale()
|
||||
|
||||
const servedByEngine = typeof window !== "undefined" && window.location.port !== "5173" && window.location.protocol.startsWith("http")
|
||||
const defaultBase = servedByEngine ? `${window.location.origin}/v1` : "http://127.0.0.1:8000/v1"
|
||||
const [baseUrl, setBaseUrl] = useState(() => {
|
||||
const saved = stored(localStorage, "colibri.baseUrl", defaultBase)
|
||||
// migrate: a stored FACTORY default pointing at another origin would trip CORS
|
||||
// when the page is engine-served — upgrade it to same-origin once.
|
||||
if (servedByEngine && saved === "http://127.0.0.1:8000/v1" && defaultBase !== saved) return defaultBase
|
||||
return saved
|
||||
})
|
||||
@@ -120,7 +119,7 @@ export default function App() {
|
||||
const result = await getHealth(baseUrl, apiKey)
|
||||
if (!disposed) { setHealth(result); setHealthError("") }
|
||||
} catch (cause) {
|
||||
if (!disposed) setHealthError(cause instanceof Error ? cause.message : "Runtime metrics unavailable")
|
||||
if (!disposed) setHealthError(cause instanceof Error ? cause.message : "status.runtimeUnavailable")
|
||||
}
|
||||
}
|
||||
const timer = window.setInterval(() => void poll(), 5000)
|
||||
@@ -155,13 +154,13 @@ export default function App() {
|
||||
} catch (cause) {
|
||||
if (!controller.signal.aborted) {
|
||||
setHealth(null)
|
||||
setHealthError(cause instanceof Error ? cause.message : "Runtime metrics unavailable")
|
||||
setHealthError(cause instanceof Error ? cause.message : "status.runtimeUnavailable")
|
||||
}
|
||||
}
|
||||
} catch (cause) {
|
||||
if (controller.signal.aborted) return
|
||||
setConnected(false)
|
||||
setError(cause instanceof Error ? cause.message : "Could not reach the server.")
|
||||
setError(cause instanceof Error ? cause.message : "status.serverError")
|
||||
} finally {
|
||||
if (probeRef.current === controller) { probeRef.current = null; setConnecting(false) }
|
||||
}
|
||||
@@ -227,7 +226,7 @@ export default function App() {
|
||||
if (controller.signal.aborted) {
|
||||
updateMessages((current) => current.filter((item) => item.id !== assistant.id || item.content))
|
||||
} else {
|
||||
setError(cause instanceof Error ? cause.message : "Generation failed.")
|
||||
setError(cause instanceof Error ? cause.message : "status.generationFailed")
|
||||
updateMessages((current) => current.filter((item) => item.id !== assistant.id || item.content))
|
||||
}
|
||||
} finally {
|
||||
@@ -241,22 +240,22 @@ export default function App() {
|
||||
<aside className="sidebar">
|
||||
<div className="brand-row">
|
||||
<div className="brand-mark"><Feather className="size-5" /></div>
|
||||
<div><h1>colibrì</h1><p>local giant, tiny footprint</p></div>
|
||||
<div><h1>colibrì</h1><p>{t("brand.tagline")}</p></div>
|
||||
</div>
|
||||
|
||||
<section className="side-section">
|
||||
<div className="section-title"><Link2 className="size-3.5" /> Connection</div>
|
||||
<label>API endpoint<Input value={baseUrl} onChange={(event) => setBaseUrl(event.target.value)} /></label>
|
||||
<label>API key<div className="relative"><KeyRound className="field-icon" /><Input className="pl-9" type="password" value={apiKey} placeholder="optional" onChange={(event) => setApiKey(event.target.value)} /></div><span className="field-help">Kept in memory only · sent to this endpoint</span></label>
|
||||
<div className="section-title"><Link2 className="size-3.5" /> {t("sidebar.connection")}</div>
|
||||
<label>{t("sidebar.endpoint")}<Input value={baseUrl} onChange={(event) => setBaseUrl(event.target.value)} /></label>
|
||||
<label>{t("sidebar.apiKey")}<div className="relative"><KeyRound className="field-icon" /><Input className="pl-9" type="password" value={apiKey} placeholder={t("sidebar.apiKeyPlaceholder")} onChange={(event) => setApiKey(event.target.value)} /></div><span className="field-help">{t("sidebar.apiKeyHelp")}</span></label>
|
||||
<Button type="button" variant="secondary" onClick={connect} disabled={connecting}>
|
||||
{connecting ? <LoaderCircle className="size-4 animate-spin" /> : <RefreshCw className="size-4" />}
|
||||
Probe server
|
||||
{t("sidebar.probe")}
|
||||
</Button>
|
||||
<div className={cn("connection-state", connected && "connected")} aria-live="polite"><span />{connected ? "Engine reachable" : "Not connected"}</div>
|
||||
<div className={cn("connection-state", connected && "connected")} aria-live="polite"><span />{connected ? t("status.connected") : t("status.notConnected")}</div>
|
||||
</section>
|
||||
|
||||
<section className="side-section runtime-section" aria-live="polite">
|
||||
<div className="section-title"><Activity className="size-3.5" /> Runtime</div>
|
||||
<div className="section-title"><Activity className="size-3.5" /> {t("sidebar.runtime")}</div>
|
||||
{health?.hwinfo ? <div className="hw-panel">
|
||||
{health.hwinfo.cpu ? <div className="hw-row"><Cpu className="size-3.5" /><span>{health.hwinfo.cpu}</span></div> : null}
|
||||
{health.hwinfo.gpus > 0 ? <div className="hw-row"><MonitorDot className="size-3.5" /><span>{health.hwinfo.gpus}× GPU<small>{health.hwinfo.vram_total_gb.toFixed(0)} GB VRAM</small></span></div> : null}
|
||||
@@ -265,66 +264,74 @@ export default function App() {
|
||||
</div> : null}
|
||||
{health?.scheduler ? <>
|
||||
<div className="runtime-grid">
|
||||
<div><span>Active</span><strong>{active}<small> / {capacity}</small></strong></div>
|
||||
<div><span>Queued</span><strong>{health.scheduler.queued}<small> / {health.scheduler.max_queue}</small></strong></div>
|
||||
<div><span>Completed</span><strong>{health.scheduler.completed}</strong></div>
|
||||
<div><span>Failures</span><strong>{failures}</strong></div>
|
||||
<div><span>{t("dashboard.active")}</span><strong>{active}<small> / {capacity}</small></strong></div>
|
||||
<div><span>{t("dashboard.queued")}</span><strong>{health.scheduler.queued}<small> / {health.scheduler.max_queue}</small></strong></div>
|
||||
<div><span>{t("dashboard.completed")}</span><strong>{health.scheduler.completed}</strong></div>
|
||||
<div><span>{t("dashboard.failures")}</span><strong>{failures}</strong></div>
|
||||
</div>
|
||||
{health.tiers ? (() => {
|
||||
const t = health.tiers
|
||||
const total = Math.max(t.vram + t.ram + t.disk, 1)
|
||||
const ti = health.tiers
|
||||
const total = Math.max(ti.vram + ti.ram + ti.disk, 1)
|
||||
return <div className="tier-panel">
|
||||
<div className="tier-bar" role="img" aria-label={`Experts: ${t.vram} VRAM, ${t.ram} RAM, ${t.disk} disk`}>
|
||||
<span className="tier-vram" style={{ width: `${(100 * t.vram) / total}%` }} />
|
||||
<span className="tier-ram" style={{ width: `${(100 * t.ram) / total}%` }} />
|
||||
<span className="tier-disk" style={{ width: `${(100 * t.disk) / total}%` }} />
|
||||
<div className="tier-bar" role="img" aria-label={t("tier.ariaLabel", { vram: ti.vram, ram: ti.ram, disk: ti.disk })}>
|
||||
<span className="tier-vram" style={{ width: `${(100 * ti.vram) / total}%` }} />
|
||||
<span className="tier-ram" style={{ width: `${(100 * ti.ram) / total}%` }} />
|
||||
<span className="tier-disk" style={{ width: `${(100 * ti.disk) / total}%` }} />
|
||||
</div>
|
||||
<div className="tier-legend">
|
||||
<span><i className="tier-vram" />VRAM <strong>{t.vram.toLocaleString()}</strong><small>{t.vram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-ram" />RAM <strong>{t.ram.toLocaleString()}</strong><small>{t.ram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-disk" />Disk <strong>{t.disk.toLocaleString()}</strong></span>
|
||||
<span><i className="tier-vram" />{t("tier.vram")} <strong>{ti.vram.toLocaleString()}</strong><small>{ti.vram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-ram" />{t("tier.ram")} <strong>{ti.ram.toLocaleString()}</strong><small>{ti.ram_gb.toFixed(1)} GB</small></span>
|
||||
<span><i className="tier-disk" />{t("tier.disk")} <strong>{ti.disk.toLocaleString()}</strong></span>
|
||||
</div>
|
||||
</div>
|
||||
})() : null}
|
||||
{totalTokens.prompt + totalTokens.completion > 0 ? <div className="session-stats">
|
||||
<span><Database className="size-3" /> Session: <strong>{totalTokens.prompt.toLocaleString()}</strong> prompt + <strong>{totalTokens.completion.toLocaleString()}</strong> completion</span>
|
||||
<span><Database className="size-3" /> {t("dashboard.session")} <strong>{totalTokens.prompt.toLocaleString()}</strong> {t("dashboard.prompt")} + <strong>{totalTokens.completion.toLocaleString()}</strong> {t("dashboard.completion")}</span>
|
||||
</div> : null}
|
||||
<div className="runtime-foot"><span className="runtime-dot" /> Scheduler online <code>{kvSlots} KV</code></div>
|
||||
</> : <p className="runtime-unavailable">{connected ? (healthError || "Runtime metrics unavailable") : "Probe the server to inspect runtime state."}</p>}
|
||||
<div className="runtime-foot"><span className="runtime-dot" /> {t("sidebar.schedulerOnline")} <code>{kvSlots} KV</code></div>
|
||||
</> : <p className="runtime-unavailable">{connected ? (healthError ? t(healthError) : t("status.runtimeUnavailable")) : t("sidebar.runtimeProbe")}</p>}
|
||||
</section>
|
||||
|
||||
<section className="side-section">
|
||||
<div className="section-title"><SlidersHorizontal className="size-3.5" /> Inference</div>
|
||||
<label>Model<select value={model} onChange={(event) => setModel(event.target.value)}>{models.length ? models.map((id) => <option key={id}>{id}</option>) : <option>{model}</option>}</select></label>
|
||||
{health?.kv_slots && health.kv_slots > 1 ? <label>KV session<select value={cacheSlot} onChange={(event) => setCacheSlot(Number(event.target.value))} disabled={loading}>
|
||||
{Array.from({ length: kvSlots }, (_, slot) => <option key={slot} value={slot}>Session {slot + 1}</option>)}
|
||||
</select><span className="field-help">Isolated context · conversation follows the selected slot</span></label> : null}
|
||||
<label><span className="label-line"><span>Temperature</span><code>{temperature.toFixed(1)}</code></span><input className="range" type="range" min="0" max="2" step="0.1" value={temperature} onChange={(event) => setTemperature(Number(event.target.value))} /></label>
|
||||
<label>Max output tokens<Input type="number" min={1} max={4096} value={maxTokens} onChange={(event) => { const value = Number(event.target.value); if (Number.isFinite(value)) setMaxTokens(Math.min(4096, Math.max(1, Math.round(value)))) }} /></label>
|
||||
<div className="section-title"><SlidersHorizontal className="size-3.5" /> {t("sidebar.inference")}</div>
|
||||
<label>{t("sidebar.model")}<select value={model} onChange={(event) => setModel(event.target.value)}>{models.length ? models.map((id) => <option key={id}>{id}</option>) : <option>{model}</option>}</select></label>
|
||||
{health?.kv_slots && health.kv_slots > 1 ? <label>{t("sidebar.kvSession")}<select value={cacheSlot} onChange={(event) => setCacheSlot(Number(event.target.value))} disabled={loading}>
|
||||
{Array.from({ length: kvSlots }, (_, slot) => <option key={slot} value={slot}>{t("sidebar.sessionLabel", { slot: slot + 1 })}</option>)}
|
||||
</select><span className="field-help">{t("sidebar.kvSessionHelp")}</span></label> : null}
|
||||
<label><span className="label-line"><span>{t("sidebar.temperature")}</span><code>{temperature.toFixed(1)}</code></span><input className="range" type="range" min="0" max="2" step="0.1" value={temperature} onChange={(event) => setTemperature(Number(event.target.value))} /></label>
|
||||
<label>{t("sidebar.maxTokens")}<Input type="number" min={1} max={4096} value={maxTokens} onChange={(event) => { const value = Number(event.target.value); if (Number.isFinite(value)) setMaxTokens(Math.min(4096, Math.max(1, Math.round(value)))) }} /></label>
|
||||
<button type="button" className={cn("toggle-row", thinking && "active")} aria-pressed={thinking} onClick={() => setThinking((value) => !value)}>
|
||||
<span><BrainCircuit className="size-4" /> Reasoning</span><i><b /></i>
|
||||
<span><BrainCircuit className="size-4" /> {t("sidebar.reasoning")}</span><i><b /></i>
|
||||
</button>
|
||||
</section>
|
||||
|
||||
<div className="sidebar-foot"><Cpu className="size-3.5" /><span>OpenAI-compatible transport</span></div>
|
||||
<div className="sidebar-foot">
|
||||
<div><Cpu className="size-3.5" /><span>{t("sidebar.transport")}</span></div>
|
||||
<div className="locale-switcher">
|
||||
<Globe className="size-3.5" />
|
||||
<select value={locale} onChange={(e) => setLocale(e.target.value)}>
|
||||
{locales.map((l) => <option key={l.code} value={l.code}>{l.label}</option>)}
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<main className="chat-panel">
|
||||
<header className="topbar">
|
||||
<div><span className="eyebrow">ACTIVE MODEL</span><strong>{model}</strong></div>
|
||||
<div><span className="eyebrow">{t("topbar.activeModel")}</span><strong>{model}</strong></div>
|
||||
<div className="view-tabs">
|
||||
<button className={view === "chat" ? "active" : ""} onClick={() => setView("chat")}><MessageSquareText className="size-3.5" /> Chat</button>
|
||||
<button className={view === "brain" ? "active" : ""} onClick={() => setView("brain")}><BrainCircuit className="size-3.5" /> Brain</button>
|
||||
<button className={view === "profiling" ? "active" : ""} onClick={() => setView("profiling")}><Gauge className="size-3.5" /> Profiling</button>
|
||||
<button className={view === "chat" ? "active" : ""} onClick={() => setView("chat")}><MessageSquareText className="size-3.5" /> {t("nav.chat")}</button>
|
||||
<button className={view === "brain" ? "active" : ""} onClick={() => setView("brain")}><BrainCircuit className="size-3.5" /> {t("nav.brain")}</button>
|
||||
<button className={view === "profiling" ? "active" : ""} onClick={() => setView("profiling")}><Gauge className="size-3.5" /> {t("nav.profiling")}</button>
|
||||
</div>
|
||||
<div className="top-actions">
|
||||
{loading && tokenCount > 0 ? <Badge className="badge-live"><Zap className="size-3 flash" /> {tokenCount} tokens</Badge> : null}
|
||||
{!loading && tokPerSec != null ? <Badge className="badge-speed"><Gauge className="size-3" /> {tokPerSec.toFixed(1)} tok/s</Badge> : null}
|
||||
{loading && tokenCount > 0 ? <Badge className="badge-live"><Zap className="size-3 flash" /> {t("topbar.tokens", { n: tokenCount })}</Badge> : null}
|
||||
{!loading && tokPerSec != null ? <Badge className="badge-speed"><Gauge className="size-3" /> {t("topbar.tokPerSec", { n: tokPerSec.toFixed(1) })}</Badge> : null}
|
||||
{!loading && ttft != null ? <Badge><Timer className="size-3" /> TTFT {(ttft/1000).toFixed(1)}s</Badge> : null}
|
||||
{!loading && lastRun?.usage ? <Badge><Layers className="size-3" /> {lastRun.usage.prompt_tokens}→{lastRun.usage.completion_tokens}</Badge> : null}
|
||||
{lastRun?.queueWaitMs != null ? <Badge><Clock className="size-3" /> queue {Math.round(lastRun.queueWaitMs)}ms</Badge> : null}
|
||||
<Badge><MonitorDot className="size-3" /> slot {cacheSlot + 1}</Badge>
|
||||
<Button variant="ghost" size="sm" onClick={() => { updateMessages([]); setTokPerSec(null); setTtft(null); setTokenCount(0); setTotalTokens({prompt:0,completion:0}) }} disabled={!messages.length || loading}><Trash2 className="size-3.5" /> Clear</Button>
|
||||
<Badge><MonitorDot className="size-3" /> {t("topbar.slot", { n: cacheSlot + 1 })}</Badge>
|
||||
<Button variant="ghost" size="sm" onClick={() => { updateMessages([]); setTokPerSec(null); setTtft(null); setTokenCount(0); setTotalTokens({prompt:0,completion:0}) }} disabled={!messages.length || loading}><Trash2 className="size-3.5" /> {t("topbar.clear")}</Button>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
@@ -335,11 +342,11 @@ export default function App() {
|
||||
{!messages.length ? (
|
||||
<div className="empty-state">
|
||||
<div className="orb"><Feather /></div>
|
||||
<span className="eyebrow">COLIBRÌ ENGINE</span>
|
||||
<h2>Ask the giant.<br /><em>Keep the machine yours.</em></h2>
|
||||
<p>Connect to a local colibrì server and stream responses directly from your hardware. Nothing leaves the endpoint you choose.</p>
|
||||
<span className="eyebrow">{t("hero.title")}</span>
|
||||
<h2>{t("hero.subtitle")}<br /><em>{t("hero.tagline")}</em></h2>
|
||||
<p>{t("hero.description")}</p>
|
||||
<div className="suggestions">
|
||||
{["Explain how expert routing works", "Write a small C benchmark", "Compare RAM and VRAM caching"].map((item) => <button key={item} onClick={() => setDraft(item)}>{item}<ArrowUp className="size-3.5 rotate-45" /></button>)}
|
||||
{[t("prompts.routing"), t("prompts.benchmark"), t("prompts.caching")].map((item) => <button key={item} onClick={() => setDraft(item)}>{item}<ArrowUp className="size-3.5 rotate-45" /></button>)}
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
@@ -347,7 +354,7 @@ export default function App() {
|
||||
{messages.map((item) => (
|
||||
<article key={item.id} className={cn("message", item.role)}>
|
||||
<div className="avatar">{item.role === "user" ? "Y" : <Feather className="size-4" />}</div>
|
||||
<div><div className="message-meta">{item.role === "user" ? "You" : "colibrì"}</div><div className="message-body">{item.content || <span className="typing" aria-label="Generating"><i /><i /><i /></span>}</div></div>
|
||||
<div><div className="message-meta">{item.role === "user" ? t("chat.you") : t("chat.colibri")}</div><div className="message-body">{item.content || <span className="typing" aria-label="Generating"><i /><i /><i /></span>}</div></div>
|
||||
</article>
|
||||
))}
|
||||
<div ref={bottomRef} />
|
||||
@@ -356,10 +363,10 @@ export default function App() {
|
||||
</div>
|
||||
|
||||
<div className="composer-wrap">
|
||||
{error && <div className="error-banner" role="alert">{error}</div>}
|
||||
{error && <div className="error-banner" role="alert">{t(error)}</div>}
|
||||
<div className="composer">
|
||||
<Textarea value={draft} onChange={(event) => setDraft(event.target.value)} placeholder="Message colibrì…" onKeyDown={(event) => { if (event.key === "Enter" && !event.shiftKey && !event.nativeEvent.isComposing) { event.preventDefault(); void send() } }} />
|
||||
<div className="composer-foot"><span><MessageSquareText className="size-3.5" /> Enter to send · Shift+Enter for newline</span>{loading ? <Button variant="destructive" size="icon" aria-label="Stop generation" onClick={() => abortRef.current?.abort()}><CircleStop className="size-4" /></Button> : <Button size="icon" aria-label="Send message" disabled={!canSend} onClick={() => void send()}><ArrowUp className="size-4" /></Button>}</div>
|
||||
<Textarea value={draft} onChange={(event) => setDraft(event.target.value)} placeholder={t("chat.placeholder")} onKeyDown={(event) => { if (event.key === "Enter" && !event.shiftKey && !event.nativeEvent.isComposing) { event.preventDefault(); void send() } }} />
|
||||
<div className="composer-foot"><span><MessageSquareText className="size-3.5" /> {t("chat.inputHint")}</span>{loading ? <Button variant="destructive" size="icon" aria-label={t("chat.stop")} onClick={() => abortRef.current?.abort()}><CircleStop className="size-4" /></Button> : <Button size="icon" aria-label={t("chat.send")} disabled={!canSend} onClick={() => void send()}><ArrowUp className="size-4" /></Button>}</div>
|
||||
</div>
|
||||
</div>
|
||||
</>}
|
||||
|
||||
+21
-22
@@ -2,27 +2,26 @@ import { useEffect, useRef, useState } from "react"
|
||||
import { BrainCircuit, Flame, Layers } from "lucide-react"
|
||||
|
||||
import { endpoint } from "@/lib/api"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
interface ExpertMap { rows: number; cols: number; map: string; hits: string; seq: number }
|
||||
interface AtlasEntry { affinity: Record<string, number>; entropy: number; top: string; label: string }
|
||||
|
||||
const TIER_NAME = ["Disk", "RAM", "VRAM"]
|
||||
const TIER_KEYS = ["tier.disk", "tier.ram", "tier.vram"] as const
|
||||
const TIER_RGB: [number, number, number][] = [[58, 71, 80], [90, 155, 216], [78, 214, 165]]
|
||||
|
||||
/* Layer-depth heuristic: what this region of the network tends to specialise in.
|
||||
* Honest framing — these are the depth roles observed across MoE interpretability
|
||||
* work, not per-expert ground truth (that needs co-activation analysis, #119). */
|
||||
function depthRole(row: number, rows: number, isMtp: boolean): string {
|
||||
if (isMtp) return "MTP head — drafts the next token for speculative decoding"
|
||||
function depthRoleKey(row: number, rows: number, isMtp: boolean): string {
|
||||
if (isMtp) return "brain.mtp"
|
||||
const f = row / Math.max(rows - 1, 1)
|
||||
if (f < 0.2) return "early layers — surface features: tokens, spelling, local syntax"
|
||||
if (f < 0.45) return "lower-middle — phrase structure, word relations, simple facts"
|
||||
if (f < 0.7) return "upper-middle — semantics, long-range context, reasoning steps"
|
||||
if (f < 0.9) return "late layers — planning the answer, style, coherence"
|
||||
return "final layers — output shaping: picks the actual next-token distribution"
|
||||
if (f < 0.2) return "brain.early"
|
||||
if (f < 0.45) return "brain.lowerMiddle"
|
||||
if (f < 0.7) return "brain.upperMiddle"
|
||||
if (f < 0.9) return "brain.late"
|
||||
return "brain.final"
|
||||
}
|
||||
|
||||
export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey: string; connected: boolean }) {
|
||||
const { t } = useLocale()
|
||||
const canvasRef = useRef<HTMLCanvasElement>(null)
|
||||
const wrapRef = useRef<HTMLDivElement>(null)
|
||||
const [wrapSize, setWrapSize] = useState({ w: 1200, h: 700 })
|
||||
@@ -142,18 +141,18 @@ export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey:
|
||||
return (
|
||||
<div className="brain-page">
|
||||
<div className="brain-head">
|
||||
<div className="section-title"><BrainCircuit className="size-4" /> Expert Cortex — {data ? `${data.rows} layers × ${data.cols} experts` : "waiting for engine"}</div>
|
||||
<div className="section-title"><BrainCircuit className="size-4" /> {t("brain.title")} — {data ? t("brain.layers", { rows: data.rows, cols: data.cols }) : t("brain.waiting")}</div>
|
||||
<div className="brain-legend">
|
||||
<span><i style={{ background: "#4ed6a5" }} /> VRAM {totals[2].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#5a9bd8" }} /> RAM {totals[1].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#3a4750" }} /> Disk {totals[0].toLocaleString()}</span>
|
||||
<span><Flame className="size-3" /> brightness = routing heat</span>
|
||||
<span className="brain-pulse-hint">⚡ white flash = routed this turn</span>
|
||||
<span><i style={{ background: "#4ed6a5" }} /> {t("tier.vram")} {totals[2].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#5a9bd8" }} /> {t("tier.ram")} {totals[1].toLocaleString()}</span>
|
||||
<span><i style={{ background: "#3a4750" }} /> {t("tier.disk")} {totals[0].toLocaleString()}</span>
|
||||
<span><Flame className="size-3" /> {t("brain.brightnessHint")}</span>
|
||||
<span className="brain-pulse-hint">{t("brain.flashHint")}</span>
|
||||
</div>
|
||||
</div>
|
||||
<div className="brain-canvas-wrap" ref={wrapRef}>
|
||||
<canvas ref={canvasRef} onMouseMove={onMove} onMouseLeave={() => setTip(null)} />
|
||||
{!connected && <p className="runtime-unavailable">Connect to the engine to see the cortex.</p>}
|
||||
{!connected && <p className="runtime-unavailable">{t("brain.connectHint")}</p>}
|
||||
</div>
|
||||
{tip && data && (() => {
|
||||
const isMtp = tip.row === data.rows - 1
|
||||
@@ -162,16 +161,16 @@ export function Brain({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey:
|
||||
return (
|
||||
<div className="brain-tip" style={{ left: tip.x + 14, top: tip.y + 14 }}>
|
||||
<div className="brain-tip-title"><Layers className="size-3" /> Layer {realLayer}{isMtp ? " (MTP)" : ""} · Expert {tip.col}</div>
|
||||
<div>Tier: <strong style={{ color: ["#8b9aa3", "#5a9bd8", "#4ed6a5"][tip.tier] }}>{TIER_NAME[tip.tier]}</strong></div>
|
||||
<div>Heat: <strong>{tip.heat === 0 ? "never routed" : `~2^${tip.heat} selections`}</strong></div>
|
||||
<div>Tier: <strong style={{ color: ["#8b9aa3", "#5a9bd8", "#4ed6a5"][tip.tier] }}>{t(TIER_KEYS[tip.tier])}</strong></div>
|
||||
<div>Heat: <strong>{tip.heat === 0 ? t("brain.neverRouted") : t("brain.selections", { heat: tip.heat })}</strong></div>
|
||||
{entry ? <>
|
||||
<div className={entry.label.startsWith("specialist") ? "brain-tip-spec" : undefined}>
|
||||
{entry.label.startsWith("specialist") ? `⭐ Specialist: ${entry.top}` : "Generalist"}
|
||||
{entry.label.startsWith("specialist") ? t("brain.specialist", { top: entry.top }) : t("brain.generalist")}
|
||||
<small> (entropy {entry.entropy})</small>
|
||||
</div>
|
||||
<div className="brain-tip-aff">{Object.entries(entry.affinity).sort((a, b) => b[1] - a[1]).slice(0, 3)
|
||||
.map(([c, p]) => `${c} ${Math.round(p * 100)}%`).join(" · ")}</div>
|
||||
</> : <div className="brain-tip-role">{depthRole(tip.row, data.rows, isMtp)}</div>}
|
||||
</> : <div className="brain-tip-role">{t(depthRoleKey(tip.row, data.rows, isMtp))}</div>}
|
||||
</div>
|
||||
)
|
||||
})()}
|
||||
|
||||
@@ -1,4 +1,18 @@
|
||||
import { Component, type ReactNode } from "react"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
function ErrorFallback({ error, onRetry }: { error: Error; onRetry: () => void }) {
|
||||
const { t } = useLocale()
|
||||
return (
|
||||
<div style={{ padding: "2rem", fontFamily: "ui-monospace, monospace", color: "#e5e7eb", background: "#0b0f10", minHeight: "100vh" }}>
|
||||
<h2 style={{ color: "#4ed6a5" }}>{t("error.title")}</h2>
|
||||
<p style={{ color: "#9ca3af" }}>{t("error.hint")}</p>
|
||||
<pre style={{ whiteSpace: "pre-wrap", color: "#f87171" }}>{String(error)}</pre>
|
||||
<button onClick={onRetry} style={{ marginTop: "1rem", padding: "0.5rem 1rem", background: "#1f2937", color: "#e5e7eb", border: "1px solid #374151", borderRadius: 8, cursor: "pointer" }}>{t("error.retry")}</button>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
interface State { error: Error | null; stack: string }
|
||||
export class ErrorBoundary extends Component<{ children: ReactNode }, State> {
|
||||
state: State = { error: null, stack: "" }
|
||||
@@ -9,11 +23,6 @@ export class ErrorBoundary extends Component<{ children: ReactNode }, State> {
|
||||
}
|
||||
render() {
|
||||
if (!this.state.error) return this.props.children
|
||||
return <div style={{ padding: "2rem", fontFamily: "ui-monospace, monospace", color: "#e5e7eb", background: "#0b0f10", minHeight: "100vh" }}>
|
||||
<h2 style={{ color: "#4ed6a5" }}>colibrì UI hit an error</h2>
|
||||
<p style={{ color: "#9ca3af" }}>The engine is unaffected. Try refreshing.</p>
|
||||
<pre style={{ whiteSpace: "pre-wrap", color: "#f87171" }}>{String(this.state.error)}</pre>
|
||||
<button onClick={() => this.setState({ error: null, stack: "" })} style={{ marginTop: "1rem", padding: "0.5rem 1rem", background: "#1f2937", color: "#e5e7eb", border: "1px solid #374151", borderRadius: 8, cursor: "pointer" }}>Retry</button>
|
||||
</div>
|
||||
return <ErrorFallback error={this.state.error} onRetry={() => this.setState({ error: null, stack: "" })} />
|
||||
}
|
||||
}
|
||||
|
||||
+26
-32
@@ -2,19 +2,14 @@ import { useEffect, useState } from "react"
|
||||
import { Activity, Gauge, HardDrive, Timer } from "lucide-react"
|
||||
|
||||
import { getProfile, type ProfileTurn } from "@/lib/api"
|
||||
import { useLocale } from "./i18n"
|
||||
|
||||
/* Wall-time phases stacked per turn. The order is the palette's CVD-safe slot
|
||||
* order (validated as a set on this surface) — identity never leans on colour
|
||||
* alone: segments keep 2px gaps, the legend is always shown and the table
|
||||
* carries the exact numbers. Disk *service* time is reported separately: it
|
||||
* runs on I/O threads overlapped with compute, so only the stall the compute
|
||||
* thread actually felt (I/O wait) belongs inside the wall-time stack. */
|
||||
const PHASES = [
|
||||
{ key: "expert_wait_s", name: "I/O wait", color: "#3987e5" },
|
||||
{ key: "expert_matmul_s", name: "Expert matmul", color: "#199e70" },
|
||||
{ key: "attention_s", name: "Attention", color: "#c98500" },
|
||||
{ key: "lm_head_s", name: "LM head", color: "#008300" },
|
||||
{ key: "other_s", name: "Other", color: "#9085e9" },
|
||||
{ key: "expert_wait_s", i18n: "profile.ioWait", color: "#3987e5" },
|
||||
{ key: "expert_matmul_s", i18n: "profile.expertMatmul", color: "#199e70" },
|
||||
{ key: "attention_s", i18n: "profile.attention", color: "#c98500" },
|
||||
{ key: "lm_head_s", i18n: "profile.lmHead", color: "#008300" },
|
||||
{ key: "other_s", i18n: "profile.other", color: "#9085e9" },
|
||||
] as const
|
||||
|
||||
interface Turn extends ProfileTurn { other_s: number; toks: number }
|
||||
@@ -28,8 +23,9 @@ const derive = (turn: ProfileTurn): Turn => ({
|
||||
const seconds = (value: number) => (value >= 10 ? value.toFixed(1) : value.toFixed(2)) + "s"
|
||||
|
||||
function ShareBar({ label, turns }: { label: string; turns: Turn[] }) {
|
||||
const { t } = useLocale()
|
||||
const total = turns.reduce((sum, turn) => sum + turn.wall_s, 0)
|
||||
const parts = PHASES.map((phase) => ({ ...phase, value: turns.reduce((sum, turn) => sum + turn[phase.key], 0) }))
|
||||
const parts = PHASES.map((phase) => ({ ...phase, name: t(phase.i18n), value: turns.reduce((sum, turn) => sum + turn[phase.key], 0) }))
|
||||
return (
|
||||
<div className="prof-share">
|
||||
<div className="prof-share-head"><span>{label}</span><code>{seconds(total)}</code></div>
|
||||
@@ -47,9 +43,7 @@ function ShareBar({ label, turns }: { label: string; turns: Turn[] }) {
|
||||
)
|
||||
}
|
||||
|
||||
/* Column chart over the recent turns; oldest on the left. Stacked mode draws the
|
||||
* wall-time composition, plain mode a single series (no legend — the title names it). */
|
||||
function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacked: boolean; height: number; format: (turn: Turn) => string }) {
|
||||
function TurnColumns({ turns, stacked, height, format, footLabel, footLabelOne }: { turns: Turn[]; stacked: boolean; height: number; format: (turn: Turn) => string; footLabel: string; footLabelOne: string }) {
|
||||
const [hover, setHover] = useState<number | null>(null)
|
||||
const peak = Math.max(...turns.map((turn) => (stacked ? turn.wall_s : turn.toks)), 1e-9)
|
||||
const gap = 2
|
||||
@@ -71,11 +65,10 @@ function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacke
|
||||
return h > 0.1 ? <rect key={`${index}-${phase.key}`} x={x} y={y + 0.35} width={width} height={Math.max(h - 0.7, 0.35)} fill={phase.color} opacity={hover === null || hover === index ? 1 : 0.45} /> : null
|
||||
})
|
||||
})}
|
||||
{/* hit targets bigger than the marks */}
|
||||
{turns.map((_, index) => <rect key={index} x={index * (width + gap) - gap / 2} y="0" width={width + gap} height={height} fill="transparent" onMouseEnter={() => setHover(index)} />)}
|
||||
</svg>
|
||||
<div className="prof-plot-foot">
|
||||
<span>{turns.length > 1 ? `${turns.length} turns · oldest → newest` : "1 turn"}</span>
|
||||
<span>{turns.length > 1 ? footLabel : footLabelOne}</span>
|
||||
<code>{hover !== null && turns[hover] ? format(turns[hover]) : `peak ${stacked ? seconds(peak) : peak.toFixed(1) + " tok/s"}`}</code>
|
||||
</div>
|
||||
</div>
|
||||
@@ -83,6 +76,7 @@ function TurnColumns({ turns, stacked, height, format }: { turns: Turn[]; stacke
|
||||
}
|
||||
|
||||
export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; apiKey: string; connected: boolean }) {
|
||||
const { t } = useLocale()
|
||||
const [turns, setTurns] = useState<Turn[]>([])
|
||||
|
||||
useEffect(() => {
|
||||
@@ -107,42 +101,42 @@ export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; api
|
||||
return (
|
||||
<div className="prof-page">
|
||||
<div className="prof-head">
|
||||
<div className="section-title"><Gauge className="size-4" /> Profiling — where the engine spends each turn</div>
|
||||
<div className="section-title"><Gauge className="size-4" /> {t("profile.title")}</div>
|
||||
<div className="prof-legend">
|
||||
{PHASES.map((phase) => <span key={phase.key}><i style={{ background: phase.color }} />{phase.name}</span>)}
|
||||
{PHASES.map((phase) => <span key={phase.key}><i style={{ background: phase.color }} />{t(phase.i18n)}</span>)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{!latest ? (
|
||||
<p className="runtime-unavailable">{connected ? "No profiled turns yet — send a chat message and the breakdown appears here." : "Connect to the engine to collect per-turn timings."}</p>
|
||||
<p className="runtime-unavailable">{connected ? t("profile.empty") : t("profile.connectHint")}</p>
|
||||
) : (
|
||||
<>
|
||||
<div className="prof-tiles">
|
||||
<div><span><Gauge className="size-3" /> Last turn</span><strong>{latest.toks.toFixed(1)}</strong><small>tok/s</small></div>
|
||||
<div><span><Timer className="size-3" /> Wall time</span><strong>{seconds(latest.wall_s)}</strong><small>{latest.prompt_tokens} → {latest.completion_tokens} tokens</small></div>
|
||||
<div><span><Activity className="size-3" /> Batching</span><strong>{latest.forwards > 0 ? (latest.completion_tokens / latest.forwards).toFixed(2) : "—"}</strong><small>tokens / forward</small></div>
|
||||
<div><span><HardDrive className="size-3" /> Disk service</span><strong>{seconds(latest.expert_disk_s)}</strong><small>overlapped with compute</small></div>
|
||||
<div><span><Gauge className="size-3" /> {t("profile.lastTurn")}</span><strong>{latest.toks.toFixed(1)}</strong><small>tok/s</small></div>
|
||||
<div><span><Timer className="size-3" /> {t("profile.wallTime")}</span><strong>{seconds(latest.wall_s)}</strong><small>{latest.prompt_tokens} → {latest.completion_tokens} tokens</small></div>
|
||||
<div><span><Activity className="size-3" /> {t("profile.batching")}</span><strong>{latest.forwards > 0 ? (latest.completion_tokens / latest.forwards).toFixed(2) : "—"}</strong><small>{t("profile.tokensPerForward")}</small></div>
|
||||
<div><span><HardDrive className="size-3" /> {t("profile.diskService")}</span><strong>{seconds(latest.expert_disk_s)}</strong><small>{t("profile.overlapped")}</small></div>
|
||||
</div>
|
||||
|
||||
<div className="prof-shares">
|
||||
<ShareBar label="Last turn" turns={[latest]} />
|
||||
{turns.length > 1 ? <ShareBar label={`Window · last ${turns.length} turns`} turns={turns} /> : null}
|
||||
<ShareBar label={t("profile.lastTurn")} turns={[latest]} />
|
||||
{turns.length > 1 ? <ShareBar label={t("profile.window", { n: turns.length })} turns={turns} /> : null}
|
||||
</div>
|
||||
|
||||
<div className="prof-charts">
|
||||
<div className="prof-chart">
|
||||
<div className="prof-chart-title">Throughput per turn (tok/s)</div>
|
||||
<TurnColumns turns={recent} stacked={false} height={36} format={(turn) => `${turn.toks.toFixed(1)} tok/s · ${turn.completion_tokens} tokens`} />
|
||||
<div className="prof-chart-title">{t("profile.throughputTitle")}</div>
|
||||
<TurnColumns turns={recent} stacked={false} height={36} footLabel={t("profile.turnsLabel", { n: recent.length })} footLabelOne={t("profile.oneTurn")} format={(turn) => `${turn.toks.toFixed(1)} tok/s · ${turn.completion_tokens} tokens`} />
|
||||
</div>
|
||||
<div className="prof-chart">
|
||||
<div className="prof-chart-title">Turn wall time by phase (s)</div>
|
||||
<TurnColumns turns={recent} stacked height={36} format={(turn) => `${seconds(turn.wall_s)} · ${PHASES.map((phase) => `${phase.name} ${seconds(turn[phase.key])}`).join(" · ")}`} />
|
||||
<div className="prof-chart-title">{t("profile.phaseTitle")}</div>
|
||||
<TurnColumns turns={recent} stacked height={36} footLabel={t("profile.turnsLabel", { n: recent.length })} footLabelOne={t("profile.oneTurn")} format={(turn) => `${seconds(turn.wall_s)} · ${PHASES.map((phase) => `${t(phase.i18n)} ${seconds(turn[phase.key])}`).join(" · ")}`} />
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="prof-table-wrap">
|
||||
<table className="prof-table">
|
||||
<thead><tr><th>Turn</th><th>Tokens</th><th>tok/s</th><th>Wall</th>{PHASES.map((phase) => <th key={phase.key}><i style={{ background: phase.color }} />{phase.name}</th>)}<th>Disk service</th></tr></thead>
|
||||
<thead><tr><th>{t("profile.turnCol")}</th><th>{t("profile.tokensCol")}</th><th>tok/s</th><th>{t("profile.wallCol")}</th>{PHASES.map((phase) => <th key={phase.key}><i style={{ background: phase.color }} />{t(phase.i18n)}</th>)}<th>{t("profile.diskService")}</th></tr></thead>
|
||||
<tbody>
|
||||
{recent.slice().reverse().map((turn, index) => (
|
||||
<tr key={turns.length - index}>
|
||||
@@ -156,7 +150,7 @@ export function Profiling({ baseUrl, apiKey, connected }: { baseUrl: string; api
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
{diskService > 0 ? <p className="prof-note">Disk service is time spent reading experts on I/O threads; it overlaps with compute, so only the <em>I/O wait</em> the compute thread felt counts inside the wall-time stack. With multiple KV sessions the shares describe the whole engine over the turn's window.</p> : null}
|
||||
{diskService > 0 ? <p className="prof-note">{t("profile.diskNote")}</p> : null}
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
const en: Record<string, string> = {
|
||||
// nav
|
||||
"nav.chat": "Chat",
|
||||
"nav.brain": "Brain",
|
||||
"nav.profiling": "Profiling",
|
||||
|
||||
// brand
|
||||
"brand.tagline": "local giant, tiny footprint",
|
||||
|
||||
// sidebar — connection
|
||||
"sidebar.connection": "Connection",
|
||||
"sidebar.endpoint": "API endpoint",
|
||||
"sidebar.apiKey": "API key",
|
||||
"sidebar.apiKeyPlaceholder": "optional",
|
||||
"sidebar.apiKeyHelp": "Kept in memory only · sent to this endpoint",
|
||||
"sidebar.probe": "Probe server",
|
||||
"status.connected": "Engine reachable",
|
||||
"status.notConnected": "Not connected",
|
||||
"status.runtimeUnavailable": "Runtime metrics unavailable",
|
||||
"status.serverError": "Could not reach the server.",
|
||||
"status.generationFailed": "Generation failed.",
|
||||
|
||||
// sidebar — runtime
|
||||
"sidebar.runtime": "Runtime",
|
||||
"sidebar.runtimeProbe": "Probe the server to inspect runtime state.",
|
||||
"sidebar.schedulerOnline": "Scheduler online",
|
||||
"dashboard.active": "Active",
|
||||
"dashboard.queued": "Queued",
|
||||
"dashboard.completed": "Completed",
|
||||
"dashboard.failures": "Failures",
|
||||
"dashboard.session": "Session:",
|
||||
"dashboard.prompt": "prompt",
|
||||
"dashboard.completion": "completion",
|
||||
|
||||
// sidebar — tiers
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "Disk",
|
||||
"tier.ariaLabel": "Experts: {{vram}} VRAM, {{ram}} RAM, {{disk}} disk",
|
||||
|
||||
// sidebar — inference
|
||||
"sidebar.inference": "Inference",
|
||||
"sidebar.model": "Model",
|
||||
"sidebar.kvSession": "KV session",
|
||||
"sidebar.kvSessionHelp": "Isolated context · conversation follows the selected slot",
|
||||
"sidebar.sessionLabel": "Session {{slot}}",
|
||||
"sidebar.temperature": "Temperature",
|
||||
"sidebar.maxTokens": "Max output tokens",
|
||||
"sidebar.reasoning": "Reasoning",
|
||||
"sidebar.transport": "OpenAI-compatible transport",
|
||||
|
||||
// top bar
|
||||
"topbar.activeModel": "ACTIVE MODEL",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "slot {{n}}",
|
||||
"topbar.clear": "Clear",
|
||||
|
||||
// hero / empty state
|
||||
"hero.title": "COLIBRÌ ENGINE",
|
||||
"hero.subtitle": "Ask the giant.",
|
||||
"hero.tagline": "Keep the machine yours.",
|
||||
"hero.description": "Connect to a local colibrì server and stream responses directly from your hardware. Nothing leaves the endpoint you choose.",
|
||||
"prompts.routing": "Explain how expert routing works",
|
||||
"prompts.benchmark": "Write a small C benchmark",
|
||||
"prompts.caching": "Compare RAM and VRAM caching",
|
||||
|
||||
// chat
|
||||
"chat.you": "You",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "Message colibrì…",
|
||||
"chat.inputHint": "Enter to send · Shift+Enter for newline",
|
||||
"chat.stop": "Stop generation",
|
||||
"chat.send": "Send message",
|
||||
|
||||
// brain
|
||||
"brain.title": "Expert Cortex",
|
||||
"brain.waiting": "waiting for engine",
|
||||
"brain.layers": "{{rows}} layers × {{cols}} experts",
|
||||
"brain.brightnessHint": "brightness = routing heat",
|
||||
"brain.flashHint": "⚡ white flash = routed this turn",
|
||||
"brain.connectHint": "Connect to the engine to see the cortex.",
|
||||
"brain.neverRouted": "never routed",
|
||||
"brain.selections": "~2^{{heat}} selections",
|
||||
"brain.specialist": "⭐ Specialist: {{top}}",
|
||||
"brain.generalist": "Generalist",
|
||||
"brain.mtp": "MTP head — drafts the next token for speculative decoding",
|
||||
"brain.early": "early layers — surface features: tokens, spelling, local syntax",
|
||||
"brain.lowerMiddle": "lower-middle — phrase structure, word relations, simple facts",
|
||||
"brain.upperMiddle": "upper-middle — semantics, long-range context, reasoning steps",
|
||||
"brain.late": "late layers — planning the answer, style, coherence",
|
||||
"brain.final": "final layers — output shaping: picks the actual next-token distribution",
|
||||
|
||||
// profiling
|
||||
"profile.title": "Profiling — where the engine spends each turn",
|
||||
"profile.ioWait": "I/O wait",
|
||||
"profile.expertMatmul": "Expert matmul",
|
||||
"profile.attention": "Attention",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "Other",
|
||||
"profile.empty": "No profiled turns yet — send a chat message and the breakdown appears here.",
|
||||
"profile.connectHint": "Connect to the engine to collect per-turn timings.",
|
||||
"profile.lastTurn": "Last turn",
|
||||
"profile.wallTime": "Wall time",
|
||||
"profile.batching": "Batching",
|
||||
"profile.tokensPerForward": "tokens / forward",
|
||||
"profile.diskService": "Disk service",
|
||||
"profile.overlapped": "overlapped with compute",
|
||||
"profile.window": "Window · last {{n}} turns",
|
||||
"profile.throughputTitle": "Throughput per turn (tok/s)",
|
||||
"profile.phaseTitle": "Turn wall time by phase (s)",
|
||||
"profile.turnCol": "Turn",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "Wall",
|
||||
"profile.turnsLabel": "{{n}} turns · oldest → newest",
|
||||
"profile.oneTurn": "1 turn",
|
||||
"profile.diskNote": "Disk service is time spent reading experts on I/O threads; it overlaps with compute, so only the I/O wait the compute thread felt counts inside the wall-time stack. With multiple KV sessions the shares describe the whole engine over the turn's window.",
|
||||
|
||||
// error boundary
|
||||
"error.title": "colibrì UI hit an error",
|
||||
"error.hint": "The engine is unaffected. Try refreshing.",
|
||||
"error.retry": "Retry",
|
||||
}
|
||||
|
||||
export default en
|
||||
@@ -0,0 +1,78 @@
|
||||
import { createContext, useContext, useState, useCallback, useMemo, type ReactNode } from "react"
|
||||
import { createElement } from "react"
|
||||
import en from "./en"
|
||||
import zhCN from "./zh-CN"
|
||||
import zhTW from "./zh-TW"
|
||||
import it from "./it"
|
||||
|
||||
const LOCALES = [
|
||||
{ code: "en", label: "English" },
|
||||
{ code: "zh-CN", label: "简体中文" },
|
||||
{ code: "zh-TW", label: "繁體中文" },
|
||||
{ code: "it", label: "Italiano" },
|
||||
] as const
|
||||
|
||||
const DICTS: Record<string, Record<string, string>> = {
|
||||
"en": en,
|
||||
"zh-CN": zhCN,
|
||||
"zh-TW": zhTW,
|
||||
"it": it,
|
||||
}
|
||||
|
||||
const STORAGE_KEY = "colibri-locale"
|
||||
|
||||
function detectLocale(): string {
|
||||
try {
|
||||
const saved = localStorage.getItem(STORAGE_KEY)
|
||||
if (saved && DICTS[saved]) return saved
|
||||
} catch {}
|
||||
const nav = navigator.language || ""
|
||||
if (DICTS[nav]) return nav
|
||||
const prefix = nav.split("-")[0]
|
||||
if (prefix === "zh") return nav.includes("TW") || nav.includes("Hant") ? "zh-TW" : "zh-CN"
|
||||
for (const { code } of LOCALES) if (code.startsWith(prefix)) return code
|
||||
return "en"
|
||||
}
|
||||
|
||||
function interpolate(template: string, vars?: Record<string, string | number>): string {
|
||||
if (!vars) return template
|
||||
return template.replace(/\{\{(\w+)\}\}/g, (_, key) => String(vars[key] ?? `{{${key}}}`))
|
||||
}
|
||||
|
||||
interface LocaleContext {
|
||||
locale: string
|
||||
setLocale: (code: string) => void
|
||||
t: (key: string, vars?: Record<string, string | number>) => string
|
||||
locales: readonly { code: string; label: string }[]
|
||||
}
|
||||
|
||||
const Ctx = createContext<LocaleContext>({
|
||||
locale: "en",
|
||||
setLocale: () => {},
|
||||
t: (key) => key,
|
||||
locales: LOCALES,
|
||||
})
|
||||
|
||||
export function LocaleProvider({ children }: { children: ReactNode }) {
|
||||
const [locale, setLocaleState] = useState(detectLocale)
|
||||
|
||||
const setLocale = useCallback((code: string) => {
|
||||
if (!DICTS[code]) return
|
||||
setLocaleState(code)
|
||||
try { localStorage.setItem(STORAGE_KEY, code) } catch {}
|
||||
}, [])
|
||||
|
||||
const t = useCallback((key: string, vars?: Record<string, string | number>) => {
|
||||
const dict = DICTS[locale] || en
|
||||
const template = dict[key] ?? en[key] ?? key
|
||||
return interpolate(template, vars)
|
||||
}, [locale])
|
||||
|
||||
const value = useMemo(() => ({ locale, setLocale, t, locales: LOCALES }), [locale, setLocale, t])
|
||||
|
||||
return createElement(Ctx.Provider, { value }, children)
|
||||
}
|
||||
|
||||
export function useLocale() {
|
||||
return useContext(Ctx)
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
const it: Record<string, string> = {
|
||||
"nav.chat": "Chat",
|
||||
"nav.brain": "Cervello",
|
||||
"nav.profiling": "Profiling",
|
||||
|
||||
"brand.tagline": "gigante locale, impronta minima",
|
||||
|
||||
"sidebar.connection": "Connessione",
|
||||
"sidebar.endpoint": "Endpoint API",
|
||||
"sidebar.apiKey": "Chiave API",
|
||||
"sidebar.apiKeyPlaceholder": "opzionale",
|
||||
"sidebar.apiKeyHelp": "Conservata solo in memoria · inviata a questo endpoint",
|
||||
"sidebar.probe": "Sonda il server",
|
||||
"status.connected": "Motore raggiungibile",
|
||||
"status.notConnected": "Non connesso",
|
||||
"status.runtimeUnavailable": "Metriche runtime non disponibili",
|
||||
"status.serverError": "Impossibile raggiungere il server.",
|
||||
"status.generationFailed": "Generazione fallita.",
|
||||
|
||||
"sidebar.runtime": "Runtime",
|
||||
"sidebar.runtimeProbe": "Sonda il server per ispezionare lo stato runtime.",
|
||||
"sidebar.schedulerOnline": "Scheduler online",
|
||||
"dashboard.active": "Attive",
|
||||
"dashboard.queued": "In coda",
|
||||
"dashboard.completed": "Completate",
|
||||
"dashboard.failures": "Fallite",
|
||||
"dashboard.session": "Sessione:",
|
||||
"dashboard.prompt": "prompt",
|
||||
"dashboard.completion": "completion",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "Disco",
|
||||
"tier.ariaLabel": "Expert: {{vram}} VRAM, {{ram}} RAM, {{disk}} disco",
|
||||
|
||||
"sidebar.inference": "Inferenza",
|
||||
"sidebar.model": "Modello",
|
||||
"sidebar.kvSession": "Sessione KV",
|
||||
"sidebar.kvSessionHelp": "Contesto isolato · la conversazione segue lo slot selezionato",
|
||||
"sidebar.sessionLabel": "Sessione {{slot}}",
|
||||
"sidebar.temperature": "Temperatura",
|
||||
"sidebar.maxTokens": "Token di output massimi",
|
||||
"sidebar.reasoning": "Ragionamento",
|
||||
"sidebar.transport": "Trasporto compatibile OpenAI",
|
||||
|
||||
"topbar.activeModel": "MODELLO ATTIVO",
|
||||
"topbar.tokens": "{{n}} token",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "slot {{n}}",
|
||||
"topbar.clear": "Pulisci",
|
||||
|
||||
"hero.title": "MOTORE COLIBRÌ",
|
||||
"hero.subtitle": "Interroga il gigante.",
|
||||
"hero.tagline": "La macchina resta tua.",
|
||||
"hero.description": "Connettiti a un server colibrì locale e ricevi le risposte in streaming direttamente dal tuo hardware. Nulla lascia l'endpoint che scegli.",
|
||||
"prompts.routing": "Spiega come funziona il routing degli expert",
|
||||
"prompts.benchmark": "Scrivi un piccolo benchmark in C",
|
||||
"prompts.caching": "Confronta il caching RAM e VRAM",
|
||||
|
||||
"chat.you": "Tu",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "Scrivi a colibrì…",
|
||||
"chat.inputHint": "Invio per inviare · Shift+Invio per andare a capo",
|
||||
"chat.stop": "Ferma la generazione",
|
||||
"chat.send": "Invia messaggio",
|
||||
|
||||
"brain.title": "Corteccia degli expert",
|
||||
"brain.waiting": "in attesa del motore",
|
||||
"brain.layers": "{{rows}} layer × {{cols}} expert",
|
||||
"brain.brightnessHint": "luminosità = calore di routing",
|
||||
"brain.flashHint": "⚡ flash bianco = instradato in questo turno",
|
||||
"brain.connectHint": "Connettiti al motore per vedere la corteccia.",
|
||||
"brain.neverRouted": "mai instradato",
|
||||
"brain.selections": "~2^{{heat}} selezioni",
|
||||
"brain.specialist": "⭐ Specialista: {{top}}",
|
||||
"brain.generalist": "Generalista",
|
||||
"brain.mtp": "Testa MTP — prepara il prossimo token per la decodifica speculativa",
|
||||
"brain.early": "layer iniziali — caratteristiche superficiali: token, ortografia, sintassi locale",
|
||||
"brain.lowerMiddle": "layer medio-bassi — struttura frasale, relazioni tra parole, fatti semplici",
|
||||
"brain.upperMiddle": "layer medio-alti — semantica, contesto a lungo raggio, passi di ragionamento",
|
||||
"brain.late": "layer avanzati — pianificazione della risposta, stile, coerenza",
|
||||
"brain.final": "layer finali — formazione dell'output: scelta della distribuzione next-token",
|
||||
|
||||
"profile.title": "Profiling — dove il motore spende ogni turno",
|
||||
"profile.ioWait": "Attesa I/O",
|
||||
"profile.expertMatmul": "Matmul expert",
|
||||
"profile.attention": "Attenzione",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "Altro",
|
||||
"profile.empty": "Nessun turno profilato — invia un messaggio e i dettagli appariranno qui.",
|
||||
"profile.connectHint": "Connettiti al motore per raccogliere i tempi per turno.",
|
||||
"profile.lastTurn": "Ultimo turno",
|
||||
"profile.wallTime": "Tempo totale",
|
||||
"profile.batching": "Batching",
|
||||
"profile.tokensPerForward": "token / forward",
|
||||
"profile.diskService": "Servizio disco",
|
||||
"profile.overlapped": "sovrapposto al calcolo",
|
||||
"profile.window": "Finestra · ultimi {{n}} turni",
|
||||
"profile.throughputTitle": "Throughput per turno (tok/s)",
|
||||
"profile.phaseTitle": "Tempo per turno per fase (s)",
|
||||
"profile.turnCol": "Turno",
|
||||
"profile.tokensCol": "Token",
|
||||
"profile.wallCol": "Totale",
|
||||
"profile.turnsLabel": "{{n}} turni · dal meno al più recente",
|
||||
"profile.oneTurn": "1 turno",
|
||||
"profile.diskNote": "Il servizio disco è il tempo speso a leggere gli expert sui thread I/O; si sovrappone al calcolo, quindi solo l'attesa I/O effettivamente percepita dal thread di calcolo conta nella ripartizione del tempo totale. Con più sessioni KV, le quote descrivono l'intero motore nella finestra del turno.",
|
||||
|
||||
"error.title": "L'interfaccia colibrì ha riscontrato un errore",
|
||||
"error.hint": "Il motore non è stato coinvolto. Prova a ricaricare la pagina.",
|
||||
"error.retry": "Riprova",
|
||||
}
|
||||
|
||||
export default it
|
||||
@@ -0,0 +1,113 @@
|
||||
const zhCN: Record<string, string> = {
|
||||
"nav.chat": "对话",
|
||||
"nav.brain": "大脑",
|
||||
"nav.profiling": "性能分析",
|
||||
|
||||
"brand.tagline": "本地巨人,极小足迹",
|
||||
|
||||
"sidebar.connection": "连接",
|
||||
"sidebar.endpoint": "API 端点",
|
||||
"sidebar.apiKey": "API 密钥",
|
||||
"sidebar.apiKeyPlaceholder": "可选",
|
||||
"sidebar.apiKeyHelp": "仅保存在内存中 · 发送到此端点",
|
||||
"sidebar.probe": "探测服务器",
|
||||
"status.connected": "引擎已连接",
|
||||
"status.notConnected": "未连接",
|
||||
"status.runtimeUnavailable": "运行时指标不可用",
|
||||
"status.serverError": "无法连接到服务器。",
|
||||
"status.generationFailed": "生成失败。",
|
||||
|
||||
"sidebar.runtime": "运行时",
|
||||
"sidebar.runtimeProbe": "探测服务器以查看运行时状态。",
|
||||
"sidebar.schedulerOnline": "调度器在线",
|
||||
"dashboard.active": "活跃",
|
||||
"dashboard.queued": "排队",
|
||||
"dashboard.completed": "已完成",
|
||||
"dashboard.failures": "失败",
|
||||
"dashboard.session": "会话:",
|
||||
"dashboard.prompt": "提示词",
|
||||
"dashboard.completion": "补全",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "磁盘",
|
||||
"tier.ariaLabel": "专家分布:{{vram}} VRAM、{{ram}} RAM、{{disk}} 磁盘",
|
||||
|
||||
"sidebar.inference": "推理",
|
||||
"sidebar.model": "模型",
|
||||
"sidebar.kvSession": "KV 会话",
|
||||
"sidebar.kvSessionHelp": "独立上下文 · 对话跟随所选槽位",
|
||||
"sidebar.sessionLabel": "会话 {{slot}}",
|
||||
"sidebar.temperature": "温度",
|
||||
"sidebar.maxTokens": "最大输出 token 数",
|
||||
"sidebar.reasoning": "推理模式",
|
||||
"sidebar.transport": "OpenAI 兼容协议",
|
||||
|
||||
"topbar.activeModel": "当前模型",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "槽位 {{n}}",
|
||||
"topbar.clear": "清空",
|
||||
|
||||
"hero.title": "COLIBRÌ 引擎",
|
||||
"hero.subtitle": "向巨人提问。",
|
||||
"hero.tagline": "让机器属于你。",
|
||||
"hero.description": "连接到本地 colibrì 服务器,直接从你的硬件流式获取响应。所有数据都留在你选择的端点内。",
|
||||
"prompts.routing": "解释专家路由是如何工作的",
|
||||
"prompts.benchmark": "写一个简单的 C 基准测试",
|
||||
"prompts.caching": "比较 RAM 和 VRAM 缓存",
|
||||
|
||||
"chat.you": "你",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "给 colibrì 发消息…",
|
||||
"chat.inputHint": "回车发送 · Shift+回车换行",
|
||||
"chat.stop": "停止生成",
|
||||
"chat.send": "发送消息",
|
||||
|
||||
"brain.title": "专家皮层",
|
||||
"brain.waiting": "等待引擎连接",
|
||||
"brain.layers": "{{rows}} 层 × {{cols}} 专家",
|
||||
"brain.brightnessHint": "亮度 = 路由热度",
|
||||
"brain.flashHint": "⚡ 白色闪烁 = 本轮被路由",
|
||||
"brain.connectHint": "连接引擎以查看皮层。",
|
||||
"brain.neverRouted": "从未被路由",
|
||||
"brain.selections": "约 2^{{heat}} 次选择",
|
||||
"brain.specialist": "⭐ 专精:{{top}}",
|
||||
"brain.generalist": "通用型",
|
||||
"brain.mtp": "MTP 头 — 为投机解码起草下一个 token",
|
||||
"brain.early": "早期层 — 表面特征:token、拼写、局部语法",
|
||||
"brain.lowerMiddle": "中低层 — 短语结构、词语关系、简单事实",
|
||||
"brain.upperMiddle": "中高层 — 语义、长距离上下文、推理步骤",
|
||||
"brain.late": "后期层 — 规划答案、风格、连贯性",
|
||||
"brain.final": "末尾层 — 输出成型:选择实际的 next-token 分布",
|
||||
|
||||
"profile.title": "性能分析 — 引擎每轮的时间花在哪里",
|
||||
"profile.ioWait": "I/O 等待",
|
||||
"profile.expertMatmul": "专家矩阵乘",
|
||||
"profile.attention": "注意力",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "其他",
|
||||
"profile.empty": "暂无性能数据 — 发送一条消息,分析结果将显示在这里。",
|
||||
"profile.connectHint": "连接引擎以采集每轮耗时。",
|
||||
"profile.lastTurn": "最近一轮",
|
||||
"profile.wallTime": "总耗时",
|
||||
"profile.batching": "批处理",
|
||||
"profile.tokensPerForward": "tokens / 前向",
|
||||
"profile.diskService": "磁盘服务",
|
||||
"profile.overlapped": "与计算重叠",
|
||||
"profile.window": "窗口 · 最近 {{n}} 轮",
|
||||
"profile.throughputTitle": "每轮吞吐量 (tok/s)",
|
||||
"profile.phaseTitle": "每轮各阶段耗时 (s)",
|
||||
"profile.turnCol": "轮次",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "总耗时",
|
||||
"profile.turnsLabel": "{{n}} 轮 · 从旧到新",
|
||||
"profile.oneTurn": "1 轮",
|
||||
"profile.diskNote": "磁盘服务是在 I/O 线程上读取专家的时间;它与计算重叠,因此只有计算线程实际感受到的 I/O 等待 才计入总耗时分解。多 KV 会话时,份额描述的是整个引擎在该轮窗口内的表现。",
|
||||
|
||||
"error.title": "colibrì UI 遇到错误",
|
||||
"error.hint": "引擎不受影响。请尝试刷新页面。",
|
||||
"error.retry": "重试",
|
||||
}
|
||||
|
||||
export default zhCN
|
||||
@@ -0,0 +1,113 @@
|
||||
const zhTW: Record<string, string> = {
|
||||
"nav.chat": "對話",
|
||||
"nav.brain": "大腦",
|
||||
"nav.profiling": "效能分析",
|
||||
|
||||
"brand.tagline": "本地巨人,極小足跡",
|
||||
|
||||
"sidebar.connection": "連線",
|
||||
"sidebar.endpoint": "API 端點",
|
||||
"sidebar.apiKey": "API 金鑰",
|
||||
"sidebar.apiKeyPlaceholder": "選填",
|
||||
"sidebar.apiKeyHelp": "僅保存在記憶體中 · 傳送到此端點",
|
||||
"sidebar.probe": "探測伺服器",
|
||||
"status.connected": "引擎已連線",
|
||||
"status.notConnected": "未連線",
|
||||
"status.runtimeUnavailable": "執行階段指標不可用",
|
||||
"status.serverError": "無法連線到伺服器。",
|
||||
"status.generationFailed": "生成失敗。",
|
||||
|
||||
"sidebar.runtime": "執行階段",
|
||||
"sidebar.runtimeProbe": "探測伺服器以檢視執行階段狀態。",
|
||||
"sidebar.schedulerOnline": "排程器上線",
|
||||
"dashboard.active": "進行中",
|
||||
"dashboard.queued": "排隊中",
|
||||
"dashboard.completed": "已完成",
|
||||
"dashboard.failures": "失敗",
|
||||
"dashboard.session": "工作階段:",
|
||||
"dashboard.prompt": "提示詞",
|
||||
"dashboard.completion": "補全",
|
||||
|
||||
"tier.vram": "VRAM",
|
||||
"tier.ram": "RAM",
|
||||
"tier.disk": "磁碟",
|
||||
"tier.ariaLabel": "專家分佈:{{vram}} VRAM、{{ram}} RAM、{{disk}} 磁碟",
|
||||
|
||||
"sidebar.inference": "推論",
|
||||
"sidebar.model": "模型",
|
||||
"sidebar.kvSession": "KV 工作階段",
|
||||
"sidebar.kvSessionHelp": "獨立上下文 · 對話跟隨所選插槽",
|
||||
"sidebar.sessionLabel": "工作階段 {{slot}}",
|
||||
"sidebar.temperature": "溫度",
|
||||
"sidebar.maxTokens": "最大輸出 token 數",
|
||||
"sidebar.reasoning": "推理模式",
|
||||
"sidebar.transport": "OpenAI 相容協定",
|
||||
|
||||
"topbar.activeModel": "目前模型",
|
||||
"topbar.tokens": "{{n}} tokens",
|
||||
"topbar.tokPerSec": "{{n}} tok/s",
|
||||
"topbar.slot": "插槽 {{n}}",
|
||||
"topbar.clear": "清除",
|
||||
|
||||
"hero.title": "COLIBRÌ 引擎",
|
||||
"hero.subtitle": "向巨人提問。",
|
||||
"hero.tagline": "讓機器屬於你。",
|
||||
"hero.description": "連線到本地 colibrì 伺服器,直接從你的硬體串流取得回應。所有資料都留在你選擇的端點內。",
|
||||
"prompts.routing": "解釋專家路由如何運作",
|
||||
"prompts.benchmark": "撰寫一個簡單的 C 基準測試",
|
||||
"prompts.caching": "比較 RAM 與 VRAM 快取",
|
||||
|
||||
"chat.you": "你",
|
||||
"chat.colibri": "colibrì",
|
||||
"chat.placeholder": "傳送訊息給 colibrì…",
|
||||
"chat.inputHint": "Enter 傳送 · Shift+Enter 換行",
|
||||
"chat.stop": "停止生成",
|
||||
"chat.send": "傳送訊息",
|
||||
|
||||
"brain.title": "專家皮層",
|
||||
"brain.waiting": "等待引擎連線",
|
||||
"brain.layers": "{{rows}} 層 × {{cols}} 專家",
|
||||
"brain.brightnessHint": "亮度 = 路由熱度",
|
||||
"brain.flashHint": "⚡ 白色閃爍 = 本輪被路由",
|
||||
"brain.connectHint": "連線引擎以檢視皮層。",
|
||||
"brain.neverRouted": "從未被路由",
|
||||
"brain.selections": "約 2^{{heat}} 次選擇",
|
||||
"brain.specialist": "⭐ 專精:{{top}}",
|
||||
"brain.generalist": "通用型",
|
||||
"brain.mtp": "MTP 頭 — 為推測式解碼起草下一個 token",
|
||||
"brain.early": "早期層 — 表面特徵:token、拼寫、局部語法",
|
||||
"brain.lowerMiddle": "中低層 — 片語結構、詞語關係、簡單事實",
|
||||
"brain.upperMiddle": "中高層 — 語意、長距離上下文、推理步驟",
|
||||
"brain.late": "後期層 — 規劃答案、風格、連貫性",
|
||||
"brain.final": "末尾層 — 輸出成型:選擇實際的 next-token 分佈",
|
||||
|
||||
"profile.title": "效能分析 — 引擎每輪的時間花在哪裡",
|
||||
"profile.ioWait": "I/O 等待",
|
||||
"profile.expertMatmul": "專家矩陣乘",
|
||||
"profile.attention": "注意力",
|
||||
"profile.lmHead": "LM head",
|
||||
"profile.other": "其他",
|
||||
"profile.empty": "尚無效能數據 — 傳送一則訊息,分析結果將顯示在這裡。",
|
||||
"profile.connectHint": "連線引擎以採集每輪耗時。",
|
||||
"profile.lastTurn": "最近一輪",
|
||||
"profile.wallTime": "總耗時",
|
||||
"profile.batching": "批次處理",
|
||||
"profile.tokensPerForward": "tokens / 前向",
|
||||
"profile.diskService": "磁碟服務",
|
||||
"profile.overlapped": "與運算重疊",
|
||||
"profile.window": "視窗 · 最近 {{n}} 輪",
|
||||
"profile.throughputTitle": "每輪吞吐量 (tok/s)",
|
||||
"profile.phaseTitle": "每輪各階段耗時 (s)",
|
||||
"profile.turnCol": "輪次",
|
||||
"profile.tokensCol": "Tokens",
|
||||
"profile.wallCol": "總耗時",
|
||||
"profile.turnsLabel": "{{n}} 輪 · 從舊到新",
|
||||
"profile.oneTurn": "1 輪",
|
||||
"profile.diskNote": "磁碟服務是在 I/O 執行緒上讀取專家的時間;它與運算重疊,因此只有運算執行緒實際感受到的 I/O 等待 才計入總耗時分解。多 KV 工作階段時,份額描述的是整個引擎在該輪視窗內的表現。",
|
||||
|
||||
"error.title": "colibrì UI 遇到錯誤",
|
||||
"error.hint": "引擎不受影響。請嘗試重新整理頁面。",
|
||||
"error.retry": "重試",
|
||||
}
|
||||
|
||||
export default zhTW
|
||||
+3
-1
@@ -64,7 +64,9 @@ button:focus-visible, input:focus-visible, textarea:focus-visible, select:focus-
|
||||
.toggle-row { display: flex; align-items: center; justify-content: space-between; height: 42px; padding: 0 11px; border: 1px solid var(--border); border-radius: 9px; color: #a9b4b8; background: var(--input); }
|
||||
.toggle-row > span { display: flex; align-items: center; gap: 8px; font-size: 11px; font-weight: 600; }.toggle-row.active { border-color: rgba(78,214,165,.35); color: var(--foreground); }
|
||||
.toggle-row i { width: 30px; height: 17px; padding: 2px; border-radius: 20px; background: #293136; transition: .2s; }.toggle-row i b { display: block; width: 13px; height: 13px; border-radius: 50%; background: #78858a; transition: .2s; }.toggle-row.active i { background: rgba(78,214,165,.28); }.toggle-row.active i b { transform: translateX(13px); background: var(--primary); }
|
||||
.sidebar-foot { margin-top: auto; display: flex; align-items: center; gap: 7px; color: #59666b; font-size: 10px; }
|
||||
.sidebar-foot { margin-top: auto; display: flex; flex-direction: column; gap: 6px; color: #59666b; font-size: 10px; }
|
||||
.sidebar-foot > div { display: flex; align-items: center; gap: 7px; }
|
||||
.locale-switcher select { background: transparent; border: 1px solid var(--border); border-radius: 4px; color: inherit; font-size: 10px; padding: 2px 4px; cursor: pointer; }
|
||||
|
||||
.chat-panel { min-width: 0; height: 100vh; display: grid; grid-template-rows: 72px minmax(0, 1fr) auto; }
|
||||
.topbar { display: flex; align-items: center; justify-content: space-between; padding: 0 32px; border-bottom: 1px solid var(--border); }
|
||||
|
||||
+4
-1
@@ -2,10 +2,13 @@ import { createRoot } from "react-dom/client"
|
||||
|
||||
import App from "./App"
|
||||
import { ErrorBoundary } from "./ErrorBoundary"
|
||||
import { LocaleProvider } from "./i18n"
|
||||
import "./index.css"
|
||||
|
||||
createRoot(document.getElementById("root")!).render(
|
||||
<ErrorBoundary>
|
||||
<App />
|
||||
<LocaleProvider>
|
||||
<App />
|
||||
</LocaleProvider>
|
||||
</ErrorBoundary>,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user