diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e18f1b3..72b7149 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -41,6 +41,78 @@ jobs: -Xcompiler=-Wall,-Wextra echo "CUDA syntax check passed" + windows-cuda-build: + # The Windows CUDA path has no coverage anywhere: `engine-cuda-syntax` above + # compiles with nvcc's *GCC* host on Linux, and check.yml's windows job is + # MinGW/UCRT64 CPU-only by design (#140). But nvcc on Windows requires MSVC + # as its host compiler — it does not accept MinGW — so `make cuda-dll` runs a + # toolchain nothing else in CI touches. That gap is not theoretical: every + # bug in #158 was a *build* failure on this path (MSVC rejects the GCC-style + # -Xcompiler=-Wall,-Wextra with "D8021 invalid numeric argument '/Wextra'", + # unresolvable CUDA_HOME/NVCC defaults, POSIX setenv in the kernel test), and + # #314 was CUDA_HOME with spaces — the layout the CUDA installer ships by + # default. Both classes are compile-time and need no GPU to catch. + # + # Build-only ON PURPOSE: GitHub's hosted runners have no NVIDIA device, so + # this job proves the Windows+MSVC CUDA build stays buildable, NOT that the + # kernels or the DLL loader behave on real silicon. That still needs hardware + # (see #157). Claiming otherwise would be the false confidence the + # engine-cuda-syntax comment above already warns about. + name: CUDA build (Windows, MSVC host) + # windows-2022, NOT windows-latest: the latest image now ships Visual Studio + # 18 (MSVC 14.5x), and CUDA's crt/host_config.h hard-errors on any host newer + # than VS 2022 ("Only the versions between 2017 and 2022 (inclusive) are + # supported"). That is a real constraint for every CUDA user on Windows, not + # a CI quirk — pinning tracks what the toolkit actually supports. Revisit when + # a CUDA release accepts VS 18; -allow-unsupported-compiler would only mask it. + runs-on: windows-2022 + steps: + - uses: actions/checkout@v4 + - name: MSVC environment (puts cl.exe on PATH for nvcc -ccbin) + uses: ilammy/msvc-dev-cmd@v1 + - name: Install CUDA toolkit (compiler only) + uses: Jimver/cuda-toolkit@v0.2.19 + with: + # Same pin as engine-cuda-syntax: v0.2.19's version table stops at + # 12.6.2. This installs to the default "C:\Program Files\NVIDIA GPU + # Computing Toolkit\..." path, so CUDA_HOME is space-bearing here — + # which is exactly the #314 regression this job would have caught. + cuda: '12.6.2' + method: network + # cudart as well as nvcc: unlike the Linux job (which only needs to + # *compile* backend_cuda.cu), cuda-dll links it, and on Windows the + # runtime headers/import lib ship as a separate installer component — + # with '["nvcc"]' alone this fails at `#include `. + sub-packages: '["nvcc", "cudart"]' + - uses: msys2/setup-msys2@v2 + with: + msystem: UCRT64 + update: false + # inherit: cl.exe (msvc-dev-cmd) and nvcc (cuda-toolkit) are added to + # the *Windows* PATH by the steps above; without inheriting it the + # recipe's `command -v` guards fail inside the MSYS2 shell. + path-type: inherit + install: >- + make + mingw-w64-ucrt-x86_64-gcc + - name: make cuda-dll (nvcc + MSVC host) + shell: msys2 {0} + run: | + cd c + # CUDA_ARCH is pinned: the default is `native`, which asks the driver + # what card is present — there is none here, so it must be explicit. + # sm_80 matches engine-cuda-syntax and is supported by the 12.6 pin. + make cuda-dll CUDA_ARCH=sm_80 + test -f coli_cuda.dll || { echo "cuda-dll reported success but produced no DLL" >&2; exit 1; } + echo "coli_cuda.dll built (MSVC host)" + - name: make glm CUDA_DLL=1 (host links backend_loader, not cudart) + shell: msys2 {0} + run: | + cd c + make glm CUDA_DLL=1 + test -f glm.exe || { echo "glm CUDA_DLL=1 reported success but produced no exe" >&2; exit 1; } + echo "glm.exe built against the DLL loader" + web: name: Web UI runs-on: ubuntu-latest diff --git a/README.md b/README.md index 053f9e8..797f1ca 100644 --- a/README.md +++ b/README.md @@ -2,12 +2,16 @@ colibrì — tiny engine, immense model

+

+ English · 繁體中文 +

+ **Tiny engine, immense model.** Run **GLM-5.2 (744B-parameter MoE)** on a consumer machine with ~25 GB of RAM — in pure C, with zero dependencies, by streaming experts from disk. -Colibrì is a lightweight, quality-preserving MoE runtime that treats VRAM, -RAM, and storage as one managed memory hierarchy. Insufficient fast memory may -reduce speed, but the default policy never silently changes model precision or -router semantics. +Colibrì is a lightweight, quality-preserving MoE runtime that treats VRAM, RAM, +and storage as one managed memory hierarchy. Insufficient fast memory may reduce +speed, but the default policy **never silently changes model precision or router +semantics**. ``` $ ./coli chat @@ -17,14 +21,14 @@ $ ./coli chat ◆ Ciao! 😊 Come posso aiutarti oggi? ``` - ## See it running

colibrì web dashboard — live metrics, hardware panel, expert tiers

-

The web dashboard (./coli web): a 744B model answering at 4+ tok/s end-to-end on 6× RTX 5090 — -with live token metrics, the hardware panel, and the VRAM/RAM/disk expert tiers.

+

The web dashboard (./coli web): a 744B model at 4 tok/s, TTFT 1.6 s, disk 0 — +full expert residency on 6× RTX 5090, with live token metrics, the per-turn time breakdown, +the VRAM/RAM/disk tier bar and the live mini-brain in the corner.

the Brain page — 19,456 experts as a live cortex @@ -33,619 +37,174 @@ with live token metrics, the hardware panel, and the VRAM/RAM/disk expert tiers. brightness is routing heat, and every expert routed in a turn flashes white. Hovering shows the expert's measured topic affinity.

-## Contents - -- [The idea](#the-idea) -- [See it running](#see-it-running) -- [What's implemented](#whats-implemented) -- [Honest numbers](#honest-numbers-wsl2-12-cores-25-gb-ram-nvme-via-vhdx) -- [Download the model](#download-the-model) -- [Web dashboard](#web-dashboard) -- [Got a better machine?](#got-a-better-machine-try-it--heres-what-to-expect) +

+ the Atlas page — the measured expert atlas as a 3-D galaxy +

+

The Atlas page: the measured expert atlas +as a 3-D galaxy — 13,260 characterised experts, 1,041 replicated specialists clustering by topic +(poetry, law, Chinese, SQL…). Position is measured routing affinity, not a learned embedding. Drag to spin.

## The idea -A 744B Mixture-of-Experts model activates only ~40B parameters per token — and only ~11 GB of those change from token to token (the routed experts). So: +A 744B Mixture-of-Experts model activates only ~40B parameters per token — and +only ~11 GB of those change from token to token (the routed experts): -- the **dense part** (attention, shared experts, embeddings — ~17B params) stays **resident in RAM at int4** (~9.9 GB); -- the **19,456 routed experts** (75 MoE layers × 256 experts + the MTP head, ~19 MB each at int4) live **on disk** (~370 GB) and are **streamed on demand**, with a per-layer LRU cache, an optional pinned hot-store, and the OS page cache as a free L2. +

+ only ~5.4% of parameters are active per token +

-The engine is a single C file (`c/glm.c`) plus small headers. No BLAS, no Python at runtime, no GPU required (an opt-in CUDA tier for pinned experts exists — see below). +So the model doesn't need to *fit* in fast memory — it needs to be **placed**: -## What's implemented +- the **dense part** (attention, shared experts, embeddings — ~17B params) stays + **resident in RAM at int4** (~9.9 GB); +- the **19,456 routed experts** (75 MoE layers × 256 + the MTP head, ~19 MB each + at int4) live **on disk** (~370 GB) and are **streamed on demand**, with a + per-layer LRU cache, a learned pinned hot-store, and an optional VRAM tier. -- **Faithful GLM-5.2 (`glm_moe_dsa`) forward** — validated token-exact against a `transformers` oracle (teacher-forcing 32/32, greedy 20/20 on a tiny-random model with the real architecture). -- **MLA attention** (q/kv-LoRA, interleaved partial RoPE) with **compressed KV-cache**: 576 floats/token instead of 32,768 (57× smaller — GLM-5.2 has 64 heads and no GQA). -- **DeepSeek-V3-style sigmoid router** (noaux_tc, routed_scaling_factor), shared expert, first-3-dense layers. -- **Native MTP speculative decoding** — GLM-5.2's own multi-token-prediction head (layer 78) drafts tokens that the main model verifies in one batched forward. **The head must be int8** (the converter does this by default): at int4 draft acceptance collapses to 0–4% and speculation never engages; at int8 it's 39–59% acceptance, **2.2–2.8 tokens/forward** (community-measured, [#8](https://github.com/JustVugg/colibri/issues/8)). Lossless *in exact arithmetic* — but **not byte-identical to non-speculative greedy in practice** ([#100](https://github.com/JustVugg/colibri/issues/100)). This isn't MTP-specific: colibrì's quantized integer kernels are shape-dependent, so any batched (S>1) or GPU forward rounds slightly differently from the single-token path, and int4 GLM-5.2 sits close enough to argmax ties that such a rounding change can flip a token. MTP, the CUDA expert tier, and batched prefill are three different ways to trip the same sensitivity (community-confirmed in #100: swapping only the kernel family forks greedy output on 3/5 prompts, with **zero speculation**). Every emitted token is still the argmax of a *valid* forward — the continuation stays correct — it just isn't the same stream. For byte-exact reproducibility: `DRAFT=0` (no speculation), plus `IDOT=0 COLI_CUDA=0` if you also want kernel-family/GPU independence. Under sampling, rejection sampling keeps the distribution correct. Honest caveat from the same measurement: on a **cold** cache each verified draft routes to extra experts (~660 → ~1100 expert-loads/token), so speculation can be a net *time* loss until the cache/pin warms up. -- **Grammar-forced speculative drafts** (`GRAMMAR=file.gbnf`, [#48](https://github.com/JustVugg/colibri/issues/48)) — on constrained-output workloads (JSON/NDJSON, function calling, structured extraction) the grammar itself is a third draft source: wherever it admits exactly **one** legal byte (braces, quotes, key names, enum bodies), that forced span is tokenized and injected as pre-accepted drafts with ~1.0 acceptance — no draft head, no lookup table, and it engages even with the int4 MTP head from [#8](https://github.com/JustVugg/colibri/issues/8). It never constrains sampling: forced spans are verified in the same batch-union forward as any draft, so a wrong or out-of-sync grammar cannot change the output — worst case is rejected drafts, and an adaptive guard turns the source off below 50% acceptance. Byte-level GBNF subset (literals, char classes, `| ( ) ? * +`, comments); `GRAMMAR_DRAFT=n` caps the forced span per forward (default 24). Composes with `DRAFT`/MTP, which fill the free-text gaps between forced spans. Full reference — mechanism, measured A/Bs, when it pays, prior art: [docs/grammar-draft.md](docs/grammar-draft.md). -- **True sampling** — temperature + nucleus, defaults tuned for int4 reality (0.7 / 0.90; the official 1.0 / 0.95 samples quantization noise from the tail). -- **Integer-dot kernels** (Q8_0-style int8 activations, AVX2 `maddubs`): int8 matmuls 1.4–2.5× faster (119 GFLOP/s measured), int4 1.8× in batch — routing decided per shape by measurement (int4 single-row stays f32: it measured slower). -- **MLA weight absorption** (DeepSeek trick) for decode: no per-token k/v reconstruction — the query absorbs `kv_b`, context is projected after attention. Validated exact: TF 32/32 and generation 20/20 with absorption forced everywhere. -- **Async expert readahead**: while one block of experts is being multiplied, the kernel is already reading the next (`WILLNEED`). -- **Quantization kernels**: int8 / packed int4 / packed int2, per-row scales, AVX2, dequant-on-use. Packing validated bit-identical to the int8 container. -- **DSA sparse attention** — GLM-5.2's lightning indexer, faithful to the reference `glm_moe_dsa` modeling: per-layer top-2048 causal key selection (full/shared indexer layers), auto-detected from the `out-idx-*` weights (`--indexer` converter mode, ~189 MB extracted from the FP8 repo). Validated exact: forcing the selection to keep every key reproduces dense attention token-for-token. `DSA=0` disables, `DSA_TOPK` overrides. -- **KV-cache persistence** — conversations reopen **warm** across engine restarts: serve mode appends the compressed MLA KV to `.coli_kv` after every turn (~182 KB/token, crash-safe) and resumes it at startup with zero re-prefill. Validated byte-identical to an uninterrupted session. `KVSAVE=0` disables. -- **Router-lookahead prefetch** (`PILOT=1`, experimental) — the next layer's routing is 71.6% predictable from the current layer's post-attention state (measured); a dedicated I/O thread prefetches those experts while the current layer computes. -- **Batch-union MoE**: in prefill (and MTP verification), each unique expert of the batch is read once and applied to every position that routes to it. -- **Byte-level BPE tokenizer in C** (GPT-2-style with Unicode-property regex, 320k merges). -- **RAM safety**: the expert cache is auto-sized from `MemAvailable` at startup — an honest peak projection (working set, KV, MTP row, reconstruction buffers) so the kernel OOM-killer never fires. -- **Offline FP8→int4 converter** (`c/tools/convert_fp8_to_int4.py`): downloads one shard at a time (~5 GB), dequants (128×128 block scales), requantizes to the engine's container, deletes the shard — the 756 GB FP8 checkpoint never needs to exist on disk at once. Resumable. +The engine is a single C file (`c/glm.c`) plus small headers. No BLAS, no Python +at runtime, no GPU required. -## Honest numbers (WSL2, 12 cores, 25 GB RAM, NVMe via VHDX) +## How it works -Detailed GPU experiment: [GLM-5.2 on 6x RTX 5090](docs/experiments/glm52-6x5090-2026-07-12.md) — full expert residency across VRAM+RAM reaches 6.84 tok/s single-request decode. +### The per-token path -| metric | value | -|---|---| -| model on disk (int4 container) | ~370 GB | -| resident RAM (dense, int4) | 9.9 GB | -| load time | ~30 s | -| peak RSS during chat | ~20 GB (auto-capped) | -| cold decode cost | ~11 GB disk reads/token (75 layers × 8 experts) | -| disk ceiling (this dev box's drive) | ~1 GB/s → ~0.05–0.1 tok/s cold | -| MTP speculation (int8 head) | 2.2–2.8 tok/forward measured ([#8](https://github.com/JustVugg/colibri/issues/8)) | +

+ route → union → place → overlap → learn +

-This is not fast. It is a 744B frontier-class model **answering correctly on a machine that costs less than one H100 fan**. Warm cache, pinned hot experts and MTP push the useful-response latency down considerably; the physics of the disk does the rest. +Every layer of every token walks the same five steps. The design goal is that +**placement only ever decides speed** — the router's decisions and the weights' +precision are the same whether an expert answered from VRAM or from disk. -### SSD note -Cold starts are heavy on random reads (~11 GB/token), but reads don't meaningfully wear an SSD — colibrì's streaming is read-only. The real concerns under heavy use are (1) **swap traffic** if the system runs out of RAM (writes do wear the drive — keep a sane `--ram` budget; colibrì's auto-budget is designed to stay clear of swap) and (2) **sustained thermals**: hours at full read duty cycle will heat cheaper drives. Monitor drive temperature and health. +### One memory hierarchy instead of one memory requirement -## Download the model +

+ VRAM / RAM / NVMe three-tier expert residency +

-A pre-converted **GLM-5.2 int4** model for colibrì is available on Hugging Face — **use the version with the int8 MTP heads** (matey-0's clone): +The same engine spans the whole range: on a 25 GB laptop everything streams from +disk (slow but correct); on a large host the entire expert set becomes resident +(`CUDA_EXPERT_GB=auto PIN_GB=all`) and disk drops out of the decode path +entirely. Between the tiers sits a **learning cache**: the engine records which +experts *your* workload routes to (`.coli_usage`, updated every turn) and pins +the hottest ones automatically — colibrì literally gets faster the more you use +it. On multi-socket hosts, `COLI_NUMA=1` interleaves the resident weights across +memory controllers ([#82](https://github.com/JustVugg/colibri/issues/82)). + +### Never wait for the disk twice + +Misses are expensive, so the engine spends most of its cleverness avoiding and +overlapping them: each expert's three matrices are stored adjacent and read in +one `pread`; a bounded async I/O pool (`PIPE=1`, default) loads missing experts +while resident ones compute; batched positions read each unique expert once +(**batch-union**); and a router-lookahead thread (`PILOT=1`) prefetches the next +layer's experts — routing is measurably **71.6% predictable one layer ahead**. +On GPUs, the resident pipeline (`COLI_CUDA_PIPE=2`) keeps the residual stream +on-device across layers so the CPU expert loop runs uninterrupted; on Apple +Silicon an experimental [Metal backend](docs/metal.md) does the batched expert +math on the unified-memory GPU. + +### Faithful model, compressed state + +The forward pass is validated **token-exact against a `transformers` oracle** +(teacher-forcing 32/32). MLA attention stores a compressed KV state — 576 +floats/token instead of 32,768 (**57× smaller**) — and persists it across +restarts (`.coli_kv`): conversations reopen warm with zero re-prefill, +byte-identical to an uninterrupted session. DSA sparse attention (GLM-5.2's +lightning indexer) is implemented faithfully and validated by forcing full-key +selection to reproduce dense attention exactly. + +### Speculative decoding, honestly + +GLM-5.2's native MTP head drafts tokens that the main model verifies in one +batched forward — 2.2–2.8 tokens/forward when it pays. Two hard-won rules ship +as defaults: the MTP head must be **int8** (int4 heads collapse to 0–4% +acceptance, [#8](https://github.com/JustVugg/colibri/issues/8)), and draft and +verify must compute **the same function** — `SPEC_PIN=1` pins both to one +kernel family ([#163](https://github.com/JustVugg/colibri/issues/163) is the +full forensic story). Grammar-forced drafts +([`GRAMMAR=file.gbnf`](docs/grammar-draft.md)) add ~free acceptance on +constrained JSON output. Whether speculation is a net win depends on your +cache temperature — measure, and use `DRAFT=0` when it doesn't pay. + +## What it achieves + +

+ measured decode speed by hardware class +

+ +Same engine, same int4 container — the hardware only changes where the experts +live. Highlights from the [full benchmark tables](docs/benchmarks.md): + +- **6× RTX 5090, full residency:** 5.8–6.8 tok/s decode, TTFT ~13 s + ([experiment log](docs/experiments/glm52-6x5090-2026-07-12.md)); +- **128 GB CPU-only desktop:** ~1.8 tok/s warm ([#200](https://github.com/JustVugg/colibri/issues/200)); +- **single RTX 5070 Ti laptop-class box:** 1.07 tok/s via the GPU-resident + pipeline ([#273](https://github.com/JustVugg/colibri/issues/273)); +- **25 GB dev box:** 0.05–0.1 tok/s cold — the proven floor where this project + started, and still the honest baseline. + +Quality is measured, not assumed: the int4 container's quantization cost and the +scale-granularity/rotation ablations live in +[docs/benchmarks.md](docs/benchmarks.md#quality-benchmark) and +[#108](https://github.com/JustVugg/colibri/issues/108)/[#81](https://github.com/JustVugg/colibri/issues/81). + +## Get started + +### 1. Get the model + +A pre-converted **GLM-5.2 int4** container is on Hugging Face — **use the +version with the int8 MTP heads**: **https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp** -> ⚠️ **The MTP head must be int8.** The original mirror ([jlnsrk/GLM-5.2-colibri-int4](https://huggingface.co/jlnsrk/GLM-5.2-colibri-int4)) ships **int4** MTP heads, which give **0% draft acceptance** — speculation silently never engages and you lose the ~2× MTP lever. This is the single most common "why is MTP stuck at 0%?" report ([#8](https://github.com/JustVugg/colibri/issues/8), [#102](https://github.com/JustVugg/colibri/issues/102)). The int8 head gives the measured **39–59% acceptance**. matey-0's clone above is the original int4 model with the three `out-mtp-*` files already swapped to int8 — download that one and you're done. -> -> Check what you have: `ls -l /out-mtp-*` -> · **int8 (correct):** `3527131672 / 5366238584 / 1065950496` -> · **int4 (0% acceptance):** `1765523544 / 2686077736 / 536747200` — if you see these, replace just those three files from the int8 mirror. +> ⚠️ The original mirror ships int4 MTP heads → 0% draft acceptance +> ([#8](https://github.com/JustVugg/colibri/issues/8)). Check yours: +> `ls -l /out-mtp-*` — int8 (correct) is `3527131672 / 5366238584 / 1065950496`. -Download the repository and point `COLI_MODEL` to its directory: +Or convert from the FP8 source yourself — one resumable command that never needs +the full 756 GB on disk at once: ```bash -COLI_MODEL=/path/to/GLM-5.2-colibri-int4-with-int8-mtp ./coli chat +cd c && ./setup.sh # checks gcc/OpenMP, builds, self-tests +./coli convert --model /nvme/glm52_i4 # download+convert shard by shard (python, one-time) ``` -This skips the FP8 → int4 conversion step entirely. Thanks to DatPat for the original mirror and matey-0 for the int8-head clone. - -### Quick start +### 2. Run it ```bash -cd c -./setup.sh # checks gcc/OpenMP, builds, self-tests - -# ONE command does everything model-side: downloads GLM-5.2-FP8 shard by shard -# (never needs the full 756 GB at once), converts to the int4 container, then -# converts the MTP head for speculative decoding. Resumable at any point. -# Conversion (only) needs python with: pip install torch safetensors huggingface_hub numpy -./coli convert --model /nvme/glm52_i4 # ~400 GB free on a real ext4/NVMe path - -# chat — RAM budget, expert cache and MTP are all detected automatically: -COLI_MODEL=/nvme/glm52_i4 ./coli chat +COLI_MODEL=/nvme/glm52_i4 ./coli chat # RAM budget, cache and MTP auto-detected +COLI_MODEL=/nvme/glm52_i4 ./coli plan # inspect the planned VRAM/RAM/disk placement +COLI_MODEL=/nvme/glm52_i4 ./coli doctor # read-only readiness check +./coli web --model /nvme/glm52_i4 # API + web dashboard on one port +./coli serve --model /nvme/glm52_i4 # OpenAI-compatible API only ``` -Inspect the planned storage hierarchy before loading the model: +The engine at runtime is pure C — python is only used by the one-time converter +and the optional API gateway. -```bash -COLI_MODEL=/nvme/glm52_i4 ./coli plan -COLI_MODEL=/nvme/glm52_i4 ./coli plan --gpu 0,1 --ram 128 --vram 48 --json +### 3. Go deeper -# apply the bounded plan to the normal runner -COLI_MODEL=/nvme/glm52_i4 ./coli chat --auto-tier -``` - -`coli plan` reads only safetensors headers and reports the model's exact dense/expert -footprint, runtime RAM reserve, safe expert-cache cap, and bounded VRAM hot tier. Its -versioned JSON output is intended to be shared by the CLI, API server, Web UI, and -desktop shell; it does not allocate model tensors or start inference. -`--auto-tier` applies the same plan to `chat`, `run`, `serve`, and benchmarks. It -sets the RAM budget and context immediately; the VRAM tier is enabled only when -the current `glm` binary is linked with CUDA. Explicit flags and environment -variables keep precedence over automatic values. - -Before loading the model, `coli doctor` performs a read-only readiness check and -explains whether the selected Disk/RAM/VRAM placement is runnable: - -```bash -COLI_MODEL=/nvme/glm52_i4 ./coli doctor -COLI_MODEL=/nvme/glm52_i4 ./coli doctor --gpu 0 --ram 128 --json -``` - -Doctor validates the model directory, config, tokenizer, safetensors headers, -engine executable, available RAM, requested NVIDIA devices, CUDA linkage, and the -same placement budget used by `coli plan`. It never starts `glm`, reads tensor -payloads, imports a model framework, or creates a CUDA context. The versioned JSON -report uses stable check IDs for automation. Warnings keep exit status 0; missing -requirements or an unsafe RAM projection return 1, while invalid CLI values return 2. - -The engine at runtime is pure C — python is only used by the one-time converter. - -### Windows 11 (native, no WSL) - -colibrì builds and runs natively on Windows 11 x86-64 with MinGW-w64. The port adds -a `_WIN32` compatibility layer in `c/compat.h` that maps POSIX I/O to the Windows API -(pread → ReadFile+OVERLAPPED, posix_fadvise no-op, aligned allocation, MoveFileEx rename, -GlobalMemoryStatusEx RAM detection). All platform differences stay in `compat.h`; the -engine source is unchanged. - -**Toolchain:** GCC via [winlibs](https://winlibs.com/) or MSYS2 MinGW-w64. Tested with -GCC 16.1.0 (x86_64-ucrt-posix-seh). - -```powershell -# One-time toolchain install (pick one): -scoop install mingw-winlibs # portable, no shell needed -# or: pacman -S mingw-w64-x86_64-gcc make # via MSYS2 - -# Python (needed by the `coli` CLI and API server). A fresh Windows resolves -# `python` to a Microsoft Store alias stub that opens the Store instead of -# running anything (#198) — installing the real interpreter replaces the stub: -winget install -e --id Python.Python.3.12 - -# Build (from c/ directory): -make glm.exe # GLM-5.2 engine (static, no DLL dependencies) -make olmoe.exe # OLMoE engine (same shims) -make iobench.exe # disk I/O benchmark -make test-c # run C tests -make test-python # run Python tests (requires python) - -# AVX-VNNI: Intel Alder Lake+ (and Meteor Lake+) CPUs have a 128-bit int8 -# dot-product instruction (VPDPBUSD) the engine can use for ~1.3x faster -# quantized matmul. The x86-64-v3 default (portable AVX2) compiles it out; -# build for THIS machine to enable it: -make glm.exe ARCH=native # banner prints "idot: avx-vnni" - -# Verify (tiny model, 2.4 MB): -pip install torch transformers safetensors huggingface_hub -python tools/make_glm_oracle.py # generate tiny oracle -SNAP=./glm_tiny TF=1 ./glm.exe 64 16 16 # expect "32/32 positions" - -# Run with real model: -SNAP=D:\glm52_i4 ./glm.exe 64 4 16 # batch inference -python coli chat --model D:\glm52_i4 # interactive chat -python coli serve --model D:\glm52_i4 # OpenAI-compatible API -``` - -**Warmup (overnight cache priming):** the engine's expert cache learns from -your workload. The included `warmup.ps1` script runs `coli run` in a loop with -diverse prompts to build the `.coli_usage` histogram unattended, so the next -real session starts with a large, accurate hot-expert pin. Each run saves usage -atomically on clean completion. - -```powershell -.\warmup.ps1 -Rounds 1 -Ngen 32 # ~60-90 min, durable progress -``` - -**NVIDIA GPU (optional, via runtime DLL):** on Windows the engine is built with -MinGW gcc but CUDA kernels require MSVC + nvcc. The split is clean: build the -CUDA backend into a standalone `coli_cuda.dll` (nvcc + MSVC), then the host -`glm.exe` loads it at runtime via `LoadLibrary` (`c/backend_loader.c`). The host -never links cudart directly; if the DLL is absent the engine falls back to CPU -without error. - -```powershell -# Prerequisites: CUDA Toolkit + MSVC Build Tools (cl.exe) + nvcc on PATH. -# Build the DLL from a shell with the MSVC environment set (vcvars64.bat or -# "x64 Native Tools Command Prompt for VS"): -make cuda-dll CUDA_HOME="C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v12.8" CUDA_ARCH=sm_120 - -# Build the host with the runtime loader (CUDA_DLL=1 adds -DCOLI_CUDA and -# links backend_loader.o instead of cudart): -make glm.exe CUDA_DLL=1 ARCH=native - -# Run with the GPU expert tier (8 GB VRAM budget here; scale to your free VRAM): -$env:COLI_CUDA="1"; $env:COLI_GPU="0"; $env:CUDA_EXPERT_GB="8" -python coli chat --model D:\glm52_i4 --topp 0.7 -``` - -The DLL exports 11 `extern "C"` symbols (`coli_cuda_init`, `coli_cuda_matmul`, -etc.); `backend_loader.c` resolves them via `GetProcAddress` on first use. -`ColiCudaTensor*` is opaque to the host (stored, never dereferenced), so the -MSVC-allocated struct is safe across the ABI boundary. `CUDA_ARCH` must match -your GPU's compute capability (e.g. `sm_120` for Blackwell / RTX 50-series, -`sm_89` for Ada / RTX 40-series). - -**Status:** Phase 1 complete (compiles, correct, static-linked). The Windows -GPU tier (runtime `coli_cuda.dll` via `LoadLibrary`) is implemented and -verified on RTX 50-series (sm_120). O_DIRECT (Phase 2) and full-model -validation against the transformers oracle remain separate workstreams. - -### OpenAI-compatible API - -`coli serve` keeps one model process loaded and exposes a text-only OpenAI-compatible -HTTP API. The gateway uses only the Python standard library; inference still runs in -the same dependency-free C engine. - -```bash -cd c -COLI_MODEL=/nvme/glm52_i4 COLI_API_KEY=local-secret ./coli serve \ - --host 127.0.0.1 --port 8000 --model-id glm-5.2-colibri - -curl http://127.0.0.1:8000/v1/chat/completions \ - -H 'Authorization: Bearer local-secret' \ - -H 'Content-Type: application/json' \ - -d '{ - "model": "glm-5.2-colibri", - "messages": [{"role": "user", "content": "Hello"}], - "stream": true - }' -``` - -Implemented endpoints are `GET /v1/models`, `GET /v1/models/{model}`, -`POST /v1/chat/completions`, and legacy `POST /v1/completions`. Chat and -completion requests support JSON responses, SSE streaming, usage counts, -`max_tokens`/`max_completion_tokens`, `temperature`, and `top_p`. The extension -`enable_thinking: true` enables GLM-5.2's reasoning block; the standard -`reasoning_effort` field also enables it unless set to `none`. - -The first version is deliberately text-only and serves one generation at a time: -the 744B model stays in one persistent process, so concurrent HTTP requests queue -instead of loading duplicate model copies. Tools, image/audio input, custom stop -sequences, log probabilities, and token penalties return an explicit error rather -than being silently ignored. The default bind address is localhost; set -`COLI_API_KEY` before exposing the server beyond the machine. - -Browser access from the Vite development server and Tauri local origins is enabled -by default. Repeat `--cors-origin https://your-ui.example` to allow another exact -origin, or use `--cors-origin '*'` only on a trusted local network. - -The engine owns one mutable KV context, so HTTP generation uses a bounded FIFO -admission queue instead of pretending to run unsafe parallel sequences. Configure it -with `--max-queue N` (default 8) and `--queue-timeout SECONDS` (default 300), or the -`COLI_MAX_QUEUE` / `COLI_QUEUE_TIMEOUT` environment variables. Saturated and timed-out -requests receive OpenAI-shaped HTTP 429 errors before streaming headers are sent. -`GET /health` exposes active/queued/completed/rejected counters, and successful -generation responses include `x-colibri-queue-wait-ms`. - -### Isolated KV contexts - -`coli serve --kv-slots N` allocates up to 16 independent sequence contexts. Requests -select one with the optional integer `cache_slot` field; ordinary OpenAI clients omit -it and keep the original slot 0 behavior. - -```json -{ - "model": "glm-5.2-colibri", - "messages": [{"role": "user", "content": "Continue this conversation"}], - "cache_slot": 1 -} -``` - -Each slot owns its token history, compressed MLA/DSA KV memory, MTP window, and -crash-safe persistence file (`.coli_kv`, `.coli_kv.1`, ...). The engine still executes -one sequence at a time; this establishes explicit KV ownership without pretending that -threaded HTTP is continuous batching. RAM admission accounts for every configured slot. -Use `COLI_KV_SLOTS=N` as the environment equivalent. Start with a small value: at the -default 4096-token context, every slot costs hundreds of MB. - -### Experimental Metal backend (Apple Silicon) - -On Apple Silicon the decode profile is matmul-bound, and unified memory removes the -PCIe copy tax that keeps CUDA's streaming experts on the CPU — so colibrì has an -opt-in Metal backend that runs the **routed-expert SwiGLU (batched, zero-copy from -the RAM slabs)**, the **fused decode attention** (full MLA layer in one command -buffer, S≤4), and **prefill's large GEMMs** on the GPU. Token-exact vs the CPU path. - -```bash -cd c -make glm METAL=1 # macOS only; no Xcode needed (shader compiles at runtime) -make metal-test # standalone kernel/attention correctness vs CPU reference -COLI_METAL=1 COLI_MODEL=/path/glm52_i4 ./coli chat --ram 96 -``` - -Measured on an M4 Max (128 GB, warm cache, MTP on): CPU 0.30 → Metal **0.42 tok/s (~1.4×)** -(best config adds `DIRECT=1`; ~3× vs this machine's first cold run). -Key design points: Metal's ~5 ms submit latency makes per-matmul dispatch a loss — -everything is batched into few command buffers per layer, and the resident experts' -GPU work is submitted *before* the missed experts' disk reads so I/O and compute -overlap. `COLI_METAL_GEMM_MIN` tunes the prefill GEMM row threshold (default 16). -Streaming, cache, MTP, DSA and the persistence formats are unchanged; every GPU -path falls back to the CPU per-block on any fault. Numerics are dequant→f32-MAC -(same as the CUDA tier); greedy outputs are byte-identical to the CPU engine. - -### Experimental resident CUDA backend - -colibrì includes an opt-in CUDA backend for model-resident tensors. Streaming -experts deliberately remain on the original CPU path for now: copying an expert -from NVMe to the GPU on every use would only replace the disk bottleneck with a -PCIe bottleneck. Resident quantized tensors are uploaded lazily once and reused. - -```bash -cd c -make cuda-test CUDA=1 # q8/q4/q2/f32 kernel correctness -make CUDA=1 -# optional dense-path experiment (hot experts are configured below) -COLI_CUDA=1 COLI_GPU=0 CUDA_DENSE=1 SNAP=/nvme/glm52_i4 ./glm 64 4 4 -``` - -Requirements: Linux, an NVIDIA driver, and a CUDA Toolkit under -`/usr/local/cuda` (override with `CUDA_HOME=/path/to/cuda`). `CUDA_ARCH=native` -builds for the GPU in the current machine; set an explicit architecture when -cross-compiling. Requesting CUDA with a CPU-only binary, an invalid device, or -an unavailable runtime fails at startup instead of silently falling back. - -The normal `make` build and runtime behavior are unchanged. CUDA defaults to an -expert-only accelerator. `CUDA_DENSE=1` additionally distributes resident -dense/attention projection tensors round-robin across the selected devices; -their projected footprint is reserved before the expert tier is placed. On six -RTX 5090s with a 150 GB expert tier, a warmed two-request/64-token GLM-5.2 run -improved from 1.650 to 2.157 aggregate tok/s (+30.8%) while retaining the full -expert tier. Treat this as an opt-in until the projected dense set and the 2 GB -per-device runtime reserve fit the target GPUs. -A measured `PIN` profile can promote its hottest experts into the persistent -VRAM tier while keeping the rest in RAM: - -```bash -STATS=stats.txt SNAP=/nvme/glm52_i4 ./glm 64 4 4 # collect routing frequencies first -COLI_CUDA=1 COLI_GPU=0 CUDA_EXPERT_GB=16 \ -PIN=stats.txt PIN_GB=160 SNAP=/nvme/glm52_i4 ./glm 64 4 4 -# multi-GPU expert tier, 150 GB total budget across six 32 GB devices -COLI_CUDA=1 COLI_GPUS=0,1,2,3,4,5 CUDA_EXPERT_GB=150 \ -CUDA_DENSE=1 PIN=stats.txt PIN_GB=300 RAM_GB=226 \ -SNAP=/nvme/glm52_i4 ./glm 64 4 4 -# large-RAM host: fill safe VRAM, then keep every remaining expert in RAM -COLI_CUDA=1 COLI_GPUS=0,1,2,3,4,5 CUDA_EXPERT_GB=auto \ -CUDA_DENSE=1 COLI_CUDA_ATTN=1 PIN=stats.txt PIN_GB=all RAM_GB=auto \ -SNAP=/nvme/glm52_i4 ./glm 64 4 4 -``` - -Selected experts are uploaded during startup, so capacity failures occur before -inference and the log reports their exact tensor footprint. The budget is clamped -against free VRAM after reserving the projected dense resident set and 2 GB of -runtime headroom per selected device. With `COLI_GPUS`, `CUDA_EXPERT_GB` is a -total budget across the device set; experts are assigned whole to the -least-loaded device that can hold them. Multi-GPU runs also default to -`PIN_FILL=1`: the measured hot set is placed first, then unused VRAM is filled -with zero-heat experts. `CUDA_RELEASE_HOST=1` (the multi-GPU default) releases -the RAM copy after a successful upload and reloads it from disk only if CUDA -later fails. Set either variable to `0` to restore the conservative behavior. -When host backing is released, placement is disjoint and staged: the hottest -prefix is loaded, uploaded to VRAM, and freed before the next-ranked suffix is -loaded into RAM. `PIN_GB` therefore describes the combined ranked set rather -than duplicate RAM and VRAM copies. On a 256 GB dual-socket host, moving from a -150 GB VRAM + 130 GB RAM placement to 150 GB VRAM + 150 GB RAM raised fixed-token -replay from 1.87 to 2.16 tok/s (+15.7%), reduced expert disk wait from 5.144s to -3.948s, and kept the projected RAM peak below `RAM_GB=226`. The cache cap adjusts -down automatically (54 to 40 in that run) so the larger pinned tier does not exceed -the process budget. Start lower on hosts with less available RAM. - -`CUDA_EXPERT_GB=auto` fills each selected device only up to its measured free -memory minus projected dense tensors and 2 GB of runtime headroom. `PIN_GB=all` -then loads every remaining routed expert into RAM, eliminating decode-time disk -misses when the host budget permits it. The regular `RAM_GB` guard still clamps -the per-layer working cache and rejects unsafe projections; this mode is intended -for dedicated high-memory inference hosts, not desktops running other workloads. -On a dedicated 251 GiB host with six RTX 5090s, this mode selected a 176.7 GB -VRAM expert tier and a 191.3 GB RAM tier (all 19,456 experts resident). The -mode also adapts the VRAM tier every 16 emitted tokens by swapping hot RAM -experts into existing GPU slots. A real 64-token greedy GLM-5.2 generation -measured **6.00 tok/s decode**, up from -2.20 tok/s end-to-end with the earlier 150 GB tier; expert hit rate was 100% -and disk wait was zero. Prompt prefill is reported separately. This is a -host-specific capacity result, not a portable default. - -Text-mode timing reports prefill separately from decode. The decode rate starts -after the prompt KV is built, so it is comparable to `REPLAY` throughput without -hiding time-to-first-token. -MTP speculation defaults off on CUDA because cold draft routes increase expert -traffic; an explicit `DRAFT=n` still overrides the default. - -On six RTX 5090 32 GB cards with GLM-5.2 int4, a 150 GB hot-first tier sustained -0.94 token/s over a 64-token varied prompt (87.8% expert hit rate), and reached -1.64 token/s on a warmed short prompt (99.3% hit rate). The same capacity filled -without routing heat managed only 0.29 token/s, so profile quality matters more -than raw VRAM capacity. These are single-run engineering measurements, not a -portable performance guarantee. - -Current limitations: devices use independent contexts and synchronous -host-staged activation copies—there is no P2P/NCCL dependency yet. Independent -expert groups execute concurrently across devices, but a single expert is not -sharded. The kernels are correctness-first custom kernels rather than -cuBLAS/Tensor Core kernels. - -For a reproducible backend A/B without the full checkpoint, generate the -deterministic 313M-parameter `glm_moe_dsa` fixture and run fixed-token replay: - -```bash -cd c -python tools/make_glm_bench_model.py --output /nvme/colibri-bench-medium --device cuda -python tools/benchmark_cuda_fixture.py --model /nvme/colibri-bench-medium --gpu 0 -``` - -The fixture has random weights and is not a language model. It exists only to -preserve the real MLA/MoE/streaming shapes and compare CPU streaming, dense-only -CUDA, CPU hot-store, and CUDA hot-expert execution with identical replay tokens. - -### Web interface - -`web/` contains a community-contributed browser UI (React + TypeScript, a pure -API client — it never touches the engine directly): - -```bash -cd web -npm ci && npm run dev # then point it at an OpenAI-compatible endpoint -``` - -It speaks the standard OpenAI Chat Completions protocol with SSE streaming, so it -works against the colibrì OpenAI-compatible server (in review, #21) or any other -compatible endpoint. Nothing leaves the endpoint you configure. The terminal -`coli chat` remains the first-class interface. - -Useful knobs (env or flags): `--temp T` token sampling temperature (default 0.7 + nucleus 0.90 — tuned for int4; 0 = greedy), `--topp 0.7` adaptive expert top-p (30–40% less disk), `--ngen N` max tokens per answer (`:more` in chat continues a truncated one), `--repin N` adapt RAM/VRAM hot experts every N emitted tokens, `AUTOPIN=0` disable the learning cache's auto-pin, `THINK=1` enable GLM-5.2's reasoning block, `DRAFT=n` MTP draft depth, `GRAMMAR=g.gbnf` grammar-forced drafts for constrained JSON/NDJSON output (`GRAMMAR_DRAFT=n` caps the forced span), `TF=1` teacher-forcing validation, `PILOT=1` router-lookahead disk prefetch (experimental — see below), `URING=1` Linux-only batched expert I/O (implies `PIPE=1`; also batches `PILOT_REAL`), `PIPE=0` disable the async expert-load pool (**default ON on Windows** — overlaps expert `pread` with the matmul so the CPU isn't idle waiting on the SSD; measured −18% disk service time), `RAM_GB=` claim more RAM for the expert cache than the conservative auto-detect (e.g. `RAM_GB=31` on a 32 GB host raises the cache cap and hit rate measurably), `CAP_RAISE=0` don't auto-grow the expert cache. - -### Resource policy - -`coli plan` reports the planned hot (VRAM), warm (RAM), and cold backing -(disk) tiers, the reason for each placement, and the expected bottleneck. The -default `--policy quality` and `--policy balanced` modes preserve checkpoint -quantization and router decisions unless `--topk` or `--topp` is passed; those -explicit lossy overrides print a warning and proceed. - -Auto-tier plans size OpenMP from physical cores and bind workers across cores. -Memory-bound quantized kernels can regress sharply when SMT siblings compete -for limited memory channels; explicit `OMP_*` settings always take precedence. - -```bash -coli plan --model /models/glm52_i4 --policy quality -coli run --auto-tier --policy quality "Explain MoE offloading" -# Explicit research-only router reduction: -coli run --policy experimental-fast --topk 4 "Benchmark prompt" -``` - -Disk is an immutable recovery source, not a normal decode target. If the plan -leaves cold expert bytes on disk, speed depends on cache hit rate; output -quality does not. - -Cold expert reads can use a deferred pipeline: resident RAM/VRAM experts execute -while missing experts are loaded in a bounded background I/O pool, then the -cold results join before the layer completes. The pool engages only under -`PIPE=1`; `PIPE_WORKERS=n` sets its worker count (default 8). Profiling reports -both disk service time and the smaller foreground-visible wait time so overlap -is explicit rather than credited as unexplained speedup. - -`--policy balanced` enables lossless live placement (`REPIN=64`). At safe -request boundaries, a per-layer LFRU score combines decaying session frequency -with recent access and replaces at most four sufficiently colder pinned -experts. `--policy quality` leaves live replacement off by default; `REPIN=0` -always disables it. Persistent `.coli_usage` history and session-local LFRU -state remain separate. - -For single-token q4 CPU experts, gate and up projections share one OpenMP -dispatch while retaining the same per-row AVX2/NEON arithmetic. This removes -one thread-team launch per RAM expert without activation requantization or a -lower-precision fallback. It is a stepping stone toward a persistent native -CPU expert pool, not a replacement for one. - -**The expert cache auto-sizes to your RAM** (since 2026-07-10): the engine now *raises* the LRU cap to fill your `--ram` budget instead of only lowering it. Before this fix a 128 GB machine ran with the same 8-experts/layer cache as a 16 GB one (issue #12) — **if you benchmarked colibrì before this date, rerun: your numbers were capped.** - -**Router-lookahead prefetch** (`PILOT=1`, experimental): GLM-5.2's expert routing is measurably predictable *ahead of time* — applying layer L+1's router to layer L's post-attention state recalls **71.6%** of the true top-8 (vs 41.3% for "same experts as last token"). `PILOT=1` uses this to issue next-layer expert readahead from a dedicated I/O thread while the current layer computes. On our dev box the disk is already ~80% saturated, so it measures neutral; on machines where compute and disk are balanced (like the Ryzen AI 9 in issue #12: 43% disk / 46% matmul) it should overlap real work — measurements welcome. - -**The learning cache**: the engine records which experts your usage actually routes to (`.coli_usage` next to the model, updated every turn) and at startup automatically pins the hottest ones in spare RAM. colibrì literally gets faster the more you use it. - -**Live tier adaptation** (`--repin N`, opt-in): at safe turn boundaries, a decaying -session heat map replaces cold pinned experts with hotter streamed experts. Replacement -loads the expert from disk into the existing RAM slot; GPU-backed slots immediately -refresh the same VRAM tier budget. A 25% hysteresis and a four-swap limit prevent tier -thrashing. Persistent `.coli_usage` remains the long-term signal and is not decayed. - -**Conversations reopen warm** (`.coli_kv`, since 2026-07-10): `coli chat` persists the compressed MLA KV-cache to disk after every turn (~182 KB/token, appended incrementally, crash-safe). Close the chat, reopen it tomorrow — the model still remembers the whole conversation and **zero re-prefill happens**: validated byte-identical to an uninterrupted session. `:reset` clears it, `KVSAVE=0` disables it. - -## Web dashboard - -One command serves the OpenAI-compatible API **and** the web console on the same port, then opens your browser when the engine is ready: - -```bash -cd web && npm install && npm run build # once -./coli web --model -``` - -What you get: - -- **Chat** with live metrics: a flashing token counter while generating, then tok/s, time-to-first-token, prompt→completion counts and queue wait; -- **Runtime panel**: your hardware (CPU, GPUs + VRAM, RAM, cores), the scheduler, and the live expert-tier bar — how many of the 19,456 experts sit in VRAM / RAM / disk right now; -- **Brain**: the whole model as a 76×256 cortex, one cell per expert. Colour = tier, brightness = routing heat, and the experts routed in each turn flash white and decay — you watch the model think. Hover any cell for its tier, heat and [measured topic affinity](https://github.com/JustVugg/colibri/issues/175) (specialists for code, Chinese, math, law… live in layers 11–22); -- **Profiling**: where each turn's wall time went — I/O wait vs expert matmul vs attention vs LM head — as stacked per-turn bars, plus throughput history, tokens-per-forward batching, and a table of the recent turns. The same phase timers behind the `PROFILE` line, streamed live. - -The dashboard talks to the engine over a few tiny protocol lines (`TIERS`, `EMAP`/`HITS`, `PROF`) and plain JSON endpoints — nothing heavier than the engine itself. - -## Got a better machine? Try it — here's what to expect - -colibrì was built on deliberately humble hardware (12 cores, 25 GB RAM, an older DRAM-less NVMe behind a WSL2 VHDX that measured ~1 GB/s random on *this* drive — note WSL2 VHDX is not inherently slow: a community 5090 box measured 10.5 GB/s O_DIRECT through one, [#101](https://github.com/JustVugg/colibri/issues/101)). **Every one of those constraints is a knob your machine can turn up.** The engine needs: Linux (or WSL2), macOS, or **Windows 11 natively (MinGW-w64)**; gcc with OpenMP, AVX2, ≥16 GB RAM, and the ~370 GB int4 model on a local NVMe (ext4/NTFS — never a network/9p mount). - -**How to test it, in order:** - -```bash -cd c && ./setup.sh # build + architecture self-test (expects 32/32) - -# 1) measure YOUR disk the way the engine uses it (parallel 19 MB random reads): -gcc -O2 -fopenmp iobench.c -o iobench -./iobench /path/to/glm52_i4/out-00069.safetensors 19 64 8 0 # buffered, 8 threads -./iobench /path/to/glm52_i4/out-00069.safetensors 19 64 8 1 # O_DIRECT (bypass cache) -# Caveat (#86): iobench reads a bounded ~1 GB shard, so buffered reads on a big-RAM box -# report the PAGE CACHE, not the disk. Use the O_DIRECT run (arg 1) for a true number, and -# run it on a shard you haven't touched this session (a prior buffered run caches its pages). -# On macOS there is no O_DIRECT — iobench uses F_NOCACHE, which stops *new* caching but can't -# evict pages a prior buffered run already resident-mapped, so a macOS "O_DIRECT" figure right -# after a buffered run still reads cache. Reboot or use a fresh shard for a real cold read. - -# 2) chat; watch the per-turn stats line (tok/s, expert hit-rate, RSS): -COLI_MODEL=/path/to/glm52_i4 ./coli chat - -# 3) record expert usage, then pin the hottest experts in your spare RAM: -STATS=stats.txt ./coli chat -PIN=stats.txt PIN_GB=20 ./coli chat # scale PIN_GB to your free RAM - -# 4) quality benchmarks (MMLU/HellaSwag/ARC): -./coli bench -``` - -**Back-of-envelope predictions** (decode is disk-bound: a cold token costs ~11.4 GB of expert reads; MTP speculation roughly halves the effective cost *once the cache is warm*; RAM turns cold reads into free cache hits): - -| machine | expected | +| topic | doc | |---|---| -| this dev box (WSL2 VHDX, ~1 GB/s, 25 GB RAM) | ~0.05–0.1 tok/s cold — proven baseline | -| native Linux, PCIe4 NVMe (~3–5 GB/s random), 32 GB | ~0.5–1 tok/s | -| PCIe5 NVMe or 2×NVMe RAID0 (~8–12 GB/s), 64 GB (PIN ~40 GB of hot experts) | ~2–4 tok/s | -| 128–256 GB RAM, 12 cores (hot experts cached) | ~2–4 tok/s — matmul-bound: ~80 GFLOP/token vs ~250 GFLOP/s of our AVX2 kernels | -| same RAM + 24–32 cores, or AVX-512/VNNI kernels | ~5–15 tok/s — interactive; kernel work is the multiplier | - -These are estimates, not measurements — if you run colibrì on serious hardware, **please open an issue with your numbers**: real datapoints from better machines are exactly what this project needs next. - -### Community benchmarks (measured) - -Real numbers from real machines, stock build (`setup.sh`, gcc 13), greedy decoding, `--ngen 32`, MTP active: - -| machine | disk (iobench, 19 MB × 64, 8 threads) | config | measured | -|---|---|---|---| -| Intel Core Ultra 7 270K Plus (24 threads) · WSL2 · 24 GB RAM · NVMe VHDX ([#2](https://github.com/JustVugg/colibri/issues/2)) | 1.96 GB/s buffered · 2.74 GB/s O_DIRECT | default | 0.07 tok/s · expert hit 3–4% · RSS 14.1 GB | -| 〃 | 〃 | `--topp 0.7` | **0.11 tok/s** · expert hit 11% · RSS 14.7 GB | -| Apple M5 Max (18 cores) · macOS · 128 GB unified · internal SSD ([#4](https://github.com/JustVugg/colibri/issues/4), [#5](https://github.com/JustVugg/colibri/issues/5)) | ~4 GB/s cold (the 14.2 GB/s reading was cache-influenced — see note) | default, MTP off | **1.06 tok/s** · expert hit 23% · RSS 21.8 GB | -| Apple M5 Max · macOS · 128 GB unified · 2 TB SSD · **Metal backend** ([#72](https://github.com/JustVugg/colibri/pull/72), [#87](https://github.com/JustVugg/colibri/issues/87)) | (macOS O_DIRECT figure unreliable — see note) | Metal on · `--ram 96` · 39.7 GB warm pin · MTP off | **1.83 tok/s** · expert hit 66% · warmed 1.11 → 1.83 over the run | -| 〃 · 46.9 GB pin (2.94M-selection history) · `--ram 110`, 1024-token run ([#103](https://github.com/JustVugg/colibri/issues/103)) | 〃 | Metal on (experts + attention) · MTP off | **2.06 tok/s** · hit 72.5% · coherent output · fastest datapoint yet (still on the pre-rebase Metal branch) | -| Mac Mini M4 Pro · macOS · **48 GB** unified · **Metal backend** ([#107](https://github.com/JustVugg/colibri/issues/107)) | 6.59 GB/s F_NOCACHE (fresh shard) | Metal on · `--ram 38` | **0.30 tok/s** (vs 0.18 CPU-only) — entry Apple Silicon on a third the RAM beats the 32-core 9950X row | -| Epyc 9654 ES · Linux · 4x16GB DDR5-4800-rdimm · Samsung PCIe Gen3 x4 NVME SSD | — | `MTP=1 DIRECT=1` | 0.31 tok/s · expert hit 35% · RSS 21.52 GB | -| Ryzen AI 9 HX 370 (Framework 13) · Arch Linux · 128 GB · WD SN850X, BTRFS zstd ([#12](https://github.com/JustVugg/colibri/issues/12)) | — | int8 MTP head · `--cap 32` · 46.7 GB auto-learned PIN | **0.37 tok/s** · expert hit 66% · MTP acceptance 52% (2.59 tok/fw) · RSS 105 GB | -| Ryzen 9 9950X (32 threads) · Linux · 123 GB · Crucial P3 QLC Gen3 ([#31](https://github.com/JustVugg/colibri/issues/31)) | 1.51 GB/s buffered | default, 2 runs from cold | 0.10 tok/s · hit 53% · profile 66% disk | -| 〃 same machine, model moved to a Samsung 9100 PRO PCIe 5.0 ([#31](https://github.com/JustVugg/colibri/issues/31)) | **8.81 GB/s** O_DIRECT | 〃 (usage history retained) | **0.28 tok/s** · hit 57% · profile flips: 32% disk / **57% matmul** | -| Ryzen AI Max+ 395 (Framework Desktop) · Ubuntu · 128 GB LPDDR5x · Intel Optane 905p PCIe 3.0 ([#39](https://github.com/JustVugg/colibri/issues/39)) | 3.27 GB/s buffered | int8 MTP head · fresh history (pure LRU, auto-raised cap 65) | 0.16 tok/s · hit 57% · profile 49% disk / 47% matmul | -| 〃 five runs later — learned pin 47.6 GB ([#39](https://github.com/JustVugg/colibri/issues/39)) | 〃 | `--temp 0.7 --topp 0.7` | **0.40 tok/s** · hit 71% · fastest non-Apple datapoint | -| Ryzen 7 9800X3D (16T) · WSL2 · 70 GB RAM · Samsung 9100 PRO PCIe 5.0 · RTX 5090 ([#101](https://github.com/JustVugg/colibri/issues/101)) | **10.51 GB/s** O_DIRECT | MTP off · learned pin 24 GB · hit 54% · OMP hot-team on | **0.41 tok/s** · disk-bound (36.5 s disk vs 24.0 s matmul) · **CUDA expert tier ≈ 0%** (AVX-512 CPU matches the 5090) · `--topp 0.7` → **0.52 tok/s** | -| EPYC 7443 (24C/48T, Zen3 AVX2) · Linux · **430 GB RAM** · NVMe RAID-Z1 via TrueNAS VM ([#104](https://github.com/JustVugg/colibri/issues/104)) | ~1 GB/s (VM overhead) | 77.5 GB pin · cap auto-raised to 194/layer · MTP off | **1.00 tok/s** · **hit 98%** · disk eliminated → **RAM-bandwidth + matmul bound** (no AVX-512/VNNI on Zen3) | -| Intel i5-12600K (10C/16T, AVX2) · **native Windows 11, no WSL** · 32 GB · MinGW GCC 16.1 ([#113](https://github.com/JustVugg/colibri/issues/113)) | buffered (no O_DIRECT on MinGW) | int8 MTP head · cold, small-RAM (cap ~2/layer) | **0.08 tok/s** · hit 3.7% · **MTP 57% acceptance** — first native-Windows datapoint, port validated | -| Ryzen 9 9950X3D2 (16C/32T, avx512-vnni) · native Linux · 121 GB · Samsung 9100 PRO **PCIe Gen5** · RTX 5090 (28 GB expert tier, 1475 pinned) ([#120](https://github.com/JustVugg/colibri/issues/120)) | **11.48 GB/s** O_DIRECT | `MTP=0 DIRECT=1 PIPE_WORKERS=16 PREFETCH=1` | **1.23 tok/s** · MTP-off wins disk-bound · fastest x86 datapoint yet | -| Ryzen AI Max+ 395 (Strix Halo, 16C/32T Zen5, avx512-vnni) · Arch Linux · 128 GB unified LPDDR5x · SK hynix P41 PCIe 4.0 ([#124](https://github.com/JustVugg/colibri/issues/124)) | — | `DIRECT=1 PIPE=1 --topp 0.7` · auto-pin | 0.06 cold → **1.10 tok/s** sustained · first Strix Halo / gfx1151 datapoint (unified memory: no discrete VRAM tier) | -| Intel Core Ultra 9 185H (16C/22T, avx-vnni) · **native Windows 11, no WSL** · 32 GB · Crucial P3 QLC NTFS · RTX 5070 Ti (unused) ([#128](https://github.com/JustVugg/colibri/issues/128)) | — | int8 MTP head · **with [#131](https://github.com/JustVugg/colibri/pull/131) (pipe + RAM fixes), warm cache, no GPU** | 0.03 cold → **0.5 tok/s** warm (~7-prompt warmup) · cache-warming on native Windows once the portability blockers are fixed — stock main hung on the `\r\n` READY sentinel before #131 | -| Dell Pro Max GB10 (DGX Spark: Grace 10×X925 + 10×A725, **aarch64 i8mm/sve2**) · Linux · 121 GB unified LPDDR5x · Dell OEM 4 TB NVMe · GB10 sm_121 ([#136](https://github.com/JustVugg/colibri/issues/136)) | **5.58 GB/s** O_DIRECT (NVIDIA-OEM unit in #76 was 10.74 — same platform, different SSD) | int8 MTP head · warm cache | 0.21 cold → **0.50 tok/s** warm · hit 83% · MTP 73% (3.20 tok/fw) · **matmul-bound** (matmul 130 s vs disk 58 s) — unified memory, CUDA placement tier neutral; the lever here is an i8mm compute kernel, not placement | - -Takeaways: with 24 GB of RAM the engine auto-caps the expert cache to 2 slots/layer, so decode stays cold even on a disk 2–2.7× faster than the dev box — **on small-RAM machines the RAM cap, not the disk, is the binding constraint**, exactly as the table above predicts; `--topp 0.7` alone bought a clean 1.6× end-to-end speedup. The M5 Max datapoint lands right on the table's second row: **~1 tok/s of a 744B model on a laptop SSD** — and its 14 GB/s disk shifts the bottleneck back to RAM budget and kernels. The Framework 13 rows are the cache thesis proven end-to-end on one machine: 0.29 → 0.37 tok/s (hit 28% → 66%, speculation finally engaging at 52% acceptance) just by giving the cache its RAM — int8 MTP head + a bigger cap + the learned pin. The cap part is now automatic (cap auto-raise, 2026-07-10). The 9950X pair is the cleanest bottleneck experiment yet — same machine, same history, only the disk swapped: ×5.8 disk bandwidth bought ×2.9 tokens, and the profile **flipped from 66% disk to 57% matmul**. But the crossover depends on the CPU kernel: the 9800X3D row ([#101](https://github.com/JustVugg/colibri/issues/101)) shows that with the OMP hot-team tuning on, the AVX-512 CPU matmul is fast enough that even a **10 GB/s NVMe stays disk-bound** — and there the **CUDA expert tier buys ≈ 0%**, because the CPU already matches the 5090 on expert matmul. The GPU tier earns its VRAM only when the CPU is the weak link, not by default. (Honest correction from #101: an earlier version of that report ran with the OMP tuning off, which manufactured a false matmul-bound crossover and a false +14% for CUDA — neither survived a clean re-run.) - -## Quality benchmark — help wanted - -**First measurement is in** ([#108](https://github.com/JustVugg/colibri/issues/108), thanks dnnspaul): the int4 container scored **62.5% mean acc_norm** on hellaswag/arc/mmlu (0-shot log-likelihood, n=40) — below the 85–95% published for full-precision GLM-5.2, but **the gap is not yet attributable to quantization.** Two confounds sit in the way: (1) 0-shot log-likelihood MC scoring badly underserves a *reasoning* model like GLM-5.2 (it never gets to think), so a large gap is expected even at fp16; (2) n=40 is ±14pp. The **decisive experiment** is the OLMoE fp16-vs-int4 A/B under this same harness (small enough to run both precisions) — that delta *is* the quantization cost with the scoring protocol cancelled out. Until it's run, 62.5% is a datapoint, not a verdict. - -The code is here and ready; one command runs it end to end (it auto-downloads the datasets on first use): - -```bash -cd c -pip install tokenizers datasets # in addition to the convert deps above -./coli bench # hellaswag, arc_challenge, mmlu — 40 questions each -./coli bench hellaswag --limit 200 # one task, more questions -./coli bench mmlu arc_challenge --ram 100 # pick tasks, set a RAM budget -``` - -It prints per-task accuracy (log-likelihood scoring, EleutherAI-harness style). **If you can run the OLMoE fp16-vs-int4 A/B (or a large-n GLM run), please open an issue with the numbers** — it's the measurement that turns 62.5% into either "int4 is fine, scoring artifact" or "quantization is the ceiling, grouped-scale is the priority." +| Benchmarks, community datapoints, quality measurements | [docs/benchmarks.md](docs/benchmarks.md) | +| Tuning knobs, policies, the learning cache, prefetch | [docs/tuning.md](docs/tuning.md) | +| Windows 11 native build (+ CUDA DLL) | [docs/windows.md](docs/windows.md) | +| CUDA backend, VRAM expert tier, full residency | [docs/cuda.md](docs/cuda.md) | +| Apple Silicon Metal backend | [docs/metal.md](docs/metal.md) | +| OpenAI-compatible API, KV slots, web dashboard | [docs/api.md](docs/api.md) | +| Grammar-forced drafts (structured output) | [docs/grammar-draft.md](docs/grammar-draft.md) | +| Environment variable inventory | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) | ## Supporting the project -colibrì is a one-person project, written and tested entirely on a 12-core laptop with 25 GB of RAM — the numbers above are the ceiling of what I can measure at home. If this project is useful or interesting to you and you'd like to support its development (better test hardware translates *directly* into a faster engine for everyone: real NVMe scaling data, bigger pinned caches, int2/int3 quality sweeps on real benchmarks), you can: +colibrì started as a one-person project on a 12-core laptop with 25 GB of RAM; +today its numbers come from a community of real machines. If it's useful to you: - ⭐ star the repo and share it; -- 🐛 open issues with benchmark numbers from your hardware; -- 💬 reach out via GitHub issues if you'd like to sponsor development or donate hardware. - -Every contribution, from a datapoint to a disk, moves the ceiling. +- 🐛 open issues with benchmark numbers from your hardware — datapoints move + this project more than anything else; +- 💬 reach out via GitHub issues to sponsor development or donate hardware. ## Repo layout @@ -662,20 +221,20 @@ c/ ├── tools/ offline conversion, fixtures and benchmarks ├── scripts/ long-running conversion helpers └── tests/ dependency-free C and Python tests -web/ browser UI (pure OpenAI-API client, community-maintained) +web/ browser UI (pure OpenAI-API client) desktop/ Tauri v2 desktop shell wrapping the web UI +docs/ reference docs, experiments, media ``` The runtime path intentionally stays flat and readable: `glm.c` plus its small -headers. Auxiliary Python and shell tooling is grouped separately and is never a -runtime dependency of the engine. - -From the repository root, `make`, `make check`, and `make clean` delegate to the -engine Makefile. Existing commands run from `c/` continue to work unchanged. +headers. From the repository root, `make`, `make check`, and `make clean` +delegate to the engine Makefile. ## Why "colibrì" -The hummingbird weighs a few grams, hovers in place, and visits a thousand flowers a day. This engine keeps a 744-billion-parameter giant alive on hummingbird rations: 25 GB of RAM, twelve CPU cores, and a lot of disk patience. +The hummingbird weighs a few grams, hovers in place, and visits a thousand +flowers a day. This engine keeps a 744-billion-parameter giant alive on +hummingbird rations: 25 GB of RAM, twelve CPU cores, and a lot of disk patience. ## License diff --git a/README.zh-TW.md b/README.zh-TW.md new file mode 100644 index 0000000..1662617 --- /dev/null +++ b/README.zh-TW.md @@ -0,0 +1,233 @@ +

+ colibrì——小巧引擎,龐大模型 +

+ +

+ English · 繁體中文 +

+ +**小巧引擎,龐大模型。**只要約 25 GB 記憶體,就能在消費級電腦上執行 **GLM-5.2(744B 參數的 MoE)**——以零相依套件的純 C 實作,從硬碟串流載入專家。 + +Colibrì 是一套輕量、維持品質的 MoE 執行環境,將 VRAM、RAM +與儲存裝置視為統一管理的記憶體階層。高速記憶體不足可能降低速度, +但預設策略**絕不會在未告知的情況下改變模型精度或路由語意**。 + +``` +$ ./coli chat + 🐦 colibrì v1.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU + ✓ ready in 32s · resident 9.9 GB + › ciao! + ◆ Ciao! 😊 Come posso aiutarti oggi? +``` + +## 實際運行畫面 + +

+ colibrì 網頁儀表板——即時指標、硬體面板與專家儲存層級 +

+

網頁儀表板(./coli web):744B 模型達到 4 tok/s、TTFT 1.6 秒、硬碟讀取 0—— +在 6× RTX 5090 上讓所有專家常駐,並即時顯示 token 指標、每輪耗時明細、 +VRAM/RAM/硬碟層級長條,以及角落的即時迷你大腦。

+ +

+ 大腦頁面——以即時皮質呈現 19,456 個專家 +

+

大腦(Brain)頁面:將全部 19,456 個專家呈現為活的皮質——顏色代表儲存層級, +亮度代表路由熱度,每輪被路由到的專家都會閃白。將游標停在專家上,即可查看其 +實測主題親和度

+ +

+ 圖譜頁面——以 3D 星系呈現實測專家圖譜 +

+

圖譜(Atlas)頁面:將實測專家圖譜 +呈現為 3D 星系——共 13,260 個已分析專家,其中 1,041 個可重現的專門專家會按主題聚集 +(詩歌、法律、中文、SQL……)。位置取自實測路由親和度,而非學習出的嵌入向量。拖曳即可旋轉。

+ +## 核心概念 + +744B 的專家混合(Mixture-of-Experts)模型,每個 token 只會啟用約 40B 參數—— +其中每個 token 之間會變動的只有約 11 GB(被路由到的專家): + +

+ 每個 token 只會啟用約 5.4% 的參數 +

+ +所以模型不必完整**放進**高速記憶體,而是需要正確**配置位置**: + +- **稠密部分**(注意力、共享專家、嵌入——約 17B 參數)以 int4 + **常駐 RAM**(約 9.9 GB); +- **19,456 個路由專家**(75 個 MoE 層 × 256,加上 MTP head;每個在 int4 下約 19 MB) + **存放在硬碟**(約 370 GB),並**隨需串流載入**,搭配逐層 LRU 快取、 + 會學習的熱門專家固定儲存區,以及選用的 VRAM 層級。 + +引擎由單一 C 檔(`c/glm.c`)與少量標頭檔組成。不需要 BLAS, +執行階段不需要 Python,也不需要 GPU。 + +## 運作方式 + +### 每個 token 的處理路徑 + +

+ 路由 → 聯集 → 配置 → 重疊執行 → 學習 +

+ +每個 token 的每一層都會走過相同的五個步驟。設計目標是讓 +**配置只決定速度**——無論專家是從 VRAM 或硬碟回應,路由器的決策與權重精度都完全相同。 + +### 統一記憶體階層,取代單一記憶體門檻 + +

+ VRAM/RAM/NVMe 三層專家常駐架構 +

+ +同一套引擎涵蓋完整硬體範圍:在 25 GB 筆電上,一切都從硬碟串流載入 +(慢,但結果正確);在大型主機上,則可讓整組專家常駐 +(`CUDA_EXPERT_GB=auto PIN_GB=all`),讓硬碟完全退出解碼路徑。 +兩端之間有一層**學習型快取**:引擎會記錄*你的*工作負載路由到哪些專家 +(`.coli_usage`,每輪更新),並自動固定最熱門的專家——colibrì 確實會越用越快。 +在多插槽主機上,`COLI_NUMA=1` 會將常駐權重交錯分配到各記憶體控制器 +([#82](https://github.com/JustVugg/colibri/issues/82))。 + +### 絕不為同一次硬碟讀取等待兩遍 + +快取未命中的成本很高,因此引擎大部分的巧思都用來避免或重疊處理這些讀取: +每個專家的三個矩陣相鄰儲存,並以一次 `pread` 讀取;有界非同步 I/O pool +(`PIPE=1`,預設啟用)會在常駐專家運算時載入缺少的專家;批次位置只讀取每個 +不重複專家一次(**批次聯集**);路由前瞻執行緒(`PILOT=1`)則預先載入下一層專家—— +實測顯示,路由結果提前一層時有 **71.6% 的可預測性**。 +在 GPU 上,常駐管線(`COLI_CUDA_PIPE=2`)讓殘差流跨層保留在裝置端, +使 CPU 專家迴圈不中斷;在 Apple Silicon 上,實驗性的 +[Metal 後端](docs/metal.md)會用統一記憶體 GPU 執行批次專家運算。 + +### 忠實模型,壓縮狀態 + +前向傳遞已透過 `transformers` oracle 驗證為**逐 token 完全一致** +(teacher-forcing 32/32)。MLA 注意力儲存壓縮後的 KV 狀態——每個 token 為 576 個 +浮點數,而非 32,768 個(**縮小 57×**)——並跨重新啟動持久保存 +(`.coli_kv`):對話可暖啟恢復,不需重新 prefill,結果與不中斷的工作階段 +逐位元組相同。DSA 稀疏注意力(GLM-5.2 的 lightning indexer)已忠實實作, +並透過強制選取所有 key,驗證可精確重現稠密注意力。 + +### 如實呈現推測式解碼 + +GLM-5.2 原生 MTP head 會起草 token,再由主模型以一次批次前向傳遞驗證—— +條件合適時每次 forward 可產生 2.2–2.8 個 token。兩條得來不易的規則已成為預設值: +MTP head 必須是 **int8**(int4 head 的接受率會崩落到 0–4%,見 +[#8](https://github.com/JustVugg/colibri/issues/8)),且草稿與驗證必須計算 +**相同函數**——`SPEC_PIN=1` 會把兩者固定在同一 kernel family +(完整鑑識過程見 [#163](https://github.com/JustVugg/colibri/issues/163))。 +文法強制草稿([`GRAMMAR=file.gbnf`](docs/grammar-draft.md))可在受限 JSON 輸出中, +以近乎免費的成本提高接受率。推測式解碼是否帶來淨收益取決於快取熱度——請實測, +若不划算就使用 `DRAFT=0`。 + +## 實際成果 + +

+ 各硬體等級的實測解碼速度 +

+ +同一套引擎、同一個 int4 容器——硬體只會改變專家的存放位置。 +[完整 benchmark 表格](docs/benchmarks.md)中的重點如下: + +- **6× RTX 5090,全部常駐:**解碼 5.8–6.8 tok/s,TTFT 約 13 秒 + ([實驗紀錄](docs/experiments/glm52-6x5090-2026-07-12.md)); +- **128 GB、僅使用 CPU 的桌上型電腦:**暖機後約 1.8 tok/s + ([#200](https://github.com/JustVugg/colibri/issues/200)); +- **單張 RTX 5070 Ti 的筆電級電腦:**透過 GPU 常駐管線達到 1.07 tok/s + ([#273](https://github.com/JustVugg/colibri/issues/273)); +- **25 GB 開發機:**冷啟動 0.05–0.1 tok/s——這是專案起步時已證實的下限, + 也仍是如實呈現的基準。 + +品質來自測量,而非假設:int4 容器的量化成本,以及 scale granularity/rotation +消融實驗,收錄於 [docs/benchmarks.md](docs/benchmarks.md#quality-benchmark)、 +[#108](https://github.com/JustVugg/colibri/issues/108) 與 +[#81](https://github.com/JustVugg/colibri/issues/81)。 + +## 開始使用 + +### 1. 取得模型 + +Hugging Face 上已有預先轉換的 **GLM-5.2 int4** 容器——請務必使用 +**含 int8 MTP head 的版本**: + +**https://huggingface.co/mateogrgic/GLM-5.2-colibri-int4-with-int8-mtp** + +> ⚠️ 原始鏡像使用 int4 MTP head → 草稿接受率為 0% +>([#8](https://github.com/JustVugg/colibri/issues/8))。請檢查你的版本: +> `ls -l /out-mtp-*`——正確的 int8 大小為 `3527131672 / 5366238584 / 1065950496`。 + +你也可以自行從 FP8 來源轉換——只需一條可續傳的指令,且任何時候都不需要 +在硬碟上同時存放完整的 756 GB: + +```bash +cd c && ./setup.sh # 檢查 gcc/OpenMP、建置並執行自我測試 +./coli convert --model /nvme/glm52_i4 # 逐 shard 下載並轉換(僅此一次需要 python) +``` + +### 2. 執行 + +```bash +COLI_MODEL=/nvme/glm52_i4 ./coli chat # 自動偵測 RAM 預算、快取與 MTP +COLI_MODEL=/nvme/glm52_i4 ./coli plan # 檢視規劃的 VRAM/RAM/硬碟配置 +COLI_MODEL=/nvme/glm52_i4 ./coli doctor # 唯讀就緒檢查 +./coli web --model /nvme/glm52_i4 # 在同一個連接埠提供 API 與網頁儀表板 +./coli serve --model /nvme/glm52_i4 # 僅提供 OpenAI 相容 API +``` + +引擎執行階段是純 C——python 只供單次轉換工具與選用的 API gateway 使用。 + +### 3. 深入了解 + +| 主題 | 文件 | +|---|---| +| Benchmark、社群實測數據、品質測量 | [docs/benchmarks.md](docs/benchmarks.md) | +| 調校選項、策略、學習型快取、預先載入 | [docs/tuning.md](docs/tuning.md) | +| Windows 11 原生建置(含 CUDA DLL) | [docs/windows.md](docs/windows.md) | +| CUDA 後端、VRAM 專家層級、全部常駐 | [docs/cuda.md](docs/cuda.md) | +| Apple Silicon Metal 後端 | [docs/metal.md](docs/metal.md) | +| OpenAI 相容 API、KV slots、網頁儀表板 | [docs/api.md](docs/api.md) | +| 文法強制草稿(結構化輸出) | [docs/grammar-draft.md](docs/grammar-draft.md) | +| 環境變數完整清單 | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) | + +## 支持專案 + +colibrì 最初是由一人使用 12 核心、25 GB RAM 的筆電開發; +如今它的數據來自社群中的各種真實機器。如果這個專案對你有用: + +- ⭐ 為儲存庫加星並分享; +- 🐛 以 issue 提交你的硬體 benchmark 數據——實測資料比任何其他事都更能推動專案; +- 💬 若想贊助開發或捐贈硬體,請透過 GitHub issues 聯絡。 + +## 儲存庫結構 + +``` +Makefile 根目錄建置/檢查入口 +c/ +├── glm.c 單檔 GLM 引擎 +├── st.h, tok.h, json.h 執行階段標頭檔 +├── backend_cuda.* 選用的 CUDA 層級 +├── Makefile 建置與本機檢查 +├── coli 使用者介面 CLI +├── openai_server.py OpenAI 相容 HTTP gateway +├── setup.sh 單一指令完成本機設定 +├── tools/ 離線轉換、fixtures 與 benchmarks +├── scripts/ 長時間轉換輔助工具 +└── tests/ 零相依套件的 C 與 Python 測試 +web/ 瀏覽器 UI(純 OpenAI API client) +desktop/ 包裝網頁 UI 的 Tauri v2 桌面 shell +docs/ 參考文件、實驗與媒體檔 +``` + +執行階段路徑刻意維持扁平、易讀:`glm.c` 加上少量標頭檔。 +在儲存庫根目錄執行 `make`、`make check` 與 `make clean`, +都會轉交給引擎的 Makefile。 + +## 為什麼叫做「colibrì」 + +蜂鳥只有幾公克重,能在原地懸停,並在一天內造訪上千朵花。 +這套引擎只用蜂鳥般的配給,就能讓 744B 參數的巨人運轉: +25 GB RAM、十二個 CPU 核心,以及對硬碟的大量耐心。 + +## 授權條款 + +Apache 2.0。GLM-5.2 權重由 Z.ai 以 MIT 授權發布。 diff --git a/c/Makefile b/c/Makefile index 27902cb..feb3045 100644 --- a/c/Makefile +++ b/c/Makefile @@ -166,7 +166,7 @@ else PYTHON ?= python3 endif CUDA_OBJ = -TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_grouped$(EXE) tests/test_stops$(EXE) tests/test_kv_alloc$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) +TEST_BINS = tests/test_json$(EXE) tests/test_st$(EXE) tests/test_st_pread$(EXE) tests/test_tier$(EXE) tests/test_grammar$(EXE) tests/test_schema_gbnf$(EXE) tests/test_decode_batch$(EXE) tests/test_idot$(EXE) tests/test_i4_grouped$(EXE) tests/test_stops$(EXE) tests/test_topp$(EXE) tests/test_kv_alloc$(EXE) tests/test_i4_acc512$(EXE) tests/test_compat_direct$(EXE) tests/test_dsa_select$(EXE) ifneq (,$(LINUX)) TEST_BINS += tests/test_uring$(EXE) endif @@ -293,6 +293,9 @@ iobench$(EXE): iobench.c compat.h tests/test_json$(EXE): tests/test_json.c json.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_st_pread$(EXE): tests/test_st_pread.c st.h json.h compat.h + $(CC) $(CFLAGS) -DST_PREAD_CHUNK=7 $< -o $@ $(LDFLAGS) + tests/test_st$(EXE): tests/test_st.c st.h json.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -317,6 +320,14 @@ tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c glm.c st.h uring.h json.h t tests/test_stops$(EXE): tests/test_stops.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_topp$(EXE): tests/test_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test +# gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp +tests/bench_topp$(EXE): tests/bench_topp.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c glm.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -326,6 +337,14 @@ tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_dsa_select$(EXE): tests/test_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356), +# NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select +tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + tests/test_uring$(EXE): tests/test_uring.c glm.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) diff --git a/c/coli b/c/coli index 4c27415..0dd9de3 100755 --- a/c/coli +++ b/c/coli @@ -20,7 +20,7 @@ Configuration through environment variables or flags (also valid after the subco --topp P adaptive expert top-p --topk N fixed top-k --ngen N maximum response tokens --cap N cache slots/layer """ -import os, sys, subprocess, argparse, json, time, signal, shutil, threading, re, codecs, tempfile, textwrap +import os, sys, subprocess, argparse, json, time, signal, shutil, threading, re, codecs, tempfile, textwrap, struct # The engine mmaps every shard (144+ files); macOS default RLIMIT_NOFILE is 256. if sys.platform != "win32": @@ -484,9 +484,134 @@ def cmd_run(a): e=env_for(a); e["PROMPT"]=f"[gMASK]<|user|>{prompt}<|assistant|>" sys.exit(subprocess.call([GLM, str(a.cap)], env=e)) +def server_probe(base, api_key=None, timeout=1.5): + """Is a coli serve alive at `base`? Returns its model_id, or None. + Probes /health then /v1/models — both cheap, neither touches the engine.""" + import urllib.request, urllib.error + def get(path): + req=urllib.request.Request(base.rstrip("/")+path) + if api_key: req.add_header("Authorization", f"Bearer {api_key}") + with urllib.request.urlopen(req, timeout=timeout) as r: + return json.loads(r.read().decode("utf-8","replace")) + try: + if get("/health").get("status")!="ok": return None + data=get("/v1/models").get("data") or [] + return data[0]["id"] if data else None + except Exception: + return None + +def chat_attached(a, base, model_id): + """The chat REPL over HTTP against a running `coli serve`. + + Why this exists (the cold-chat cost, measured): spawning a private engine + pays 34-136 s of resident load on EVERY start, and begins with an empty + expert cache — hit rate 4% cold vs 55% warm, a ~10x on early decode. A + resident server pays load once and keeps the LRU warm across sessions; + its KV slots reuse the conversation prefix, so a continued chat skips + re-prefill too. The engine byte-protocol stays untouched — this is plain + OpenAI SSE over localhost, stdlib only.""" + import urllib.request + print(f" {C.grn}✦ attached{C.r} {C.dim}to {base} · model {model_id} · the engine stays warm after you quit{C.r}") + print(f" {C.dim}type and press Enter · Ctrl-C stops the answer · :reset starts a new conversation · :q exits{C.r}\n") + msgs=[] + w=term_w()-4 + while True: + if TTY: + print(f" {C.dgray}╭{'─'*w}╮{C.r}") + try: msg=input(f" {C.dgray}│{C.r} {C.teal}{C.b}›{C.r} ") + except EOFError: print(); break + print(f" {C.dgray}╰{'─'*w}╯{C.r}") + else: + try: msg=input() + except EOFError: break + msg=msg.strip() + if msg in (":q",":quit","exit"): break + if not msg: continue + if msg==":reset": msgs=[]; print(f" {C.dim}✦ new conversation{C.r}\n"); continue + msgs.append({"role":"user","content":msg}) + body=json.dumps({"model":model_id,"messages":msgs,"stream":True, + "max_tokens":a.ngen}).encode() + req=urllib.request.Request(base.rstrip("/")+"/v1/chat/completions", data=body, + headers={"Content-Type":"application/json"}) + if a.api_key: req.add_header("Authorization", f"Bearer {a.api_key}") + print(f"\n {C.teal}◆ colibrì{C.r}") + sp=Spinner("thinking…"); sp.start() + md=MDStream(" "); reply=[]; first=True; t0=time.time(); interrupted=False + try: + with urllib.request.urlopen(req) as r: + for raw in r: + line=raw.decode("utf-8","replace").strip() + if not line.startswith("data: "): continue + data=line[6:] + if data=="[DONE]": break + try: ev=json.loads(data) + except ValueError: continue + for ch in ev.get("choices",[]): + d=ch.get("delta",{}) + txt=d.get("content") + if not txt: continue # ping/ruolo/reasoning: non è testo + if first: sp.stop(); first=False + md.feed(txt); reply.append(txt) + except KeyboardInterrupt: + interrupted=True # il server annulla la richiesta alla disconnessione + except OSError as e: + sp.stop() + print(f"\n {C.yel}[server unreachable: {e}]{C.r}"); break + sp.stop(); md.close() + if reply: msgs.append({"role":"assistant","content":"".join(reply)}) + else: msgs.pop() # turno vuoto: non sporcare la history + el=time.time()-t0 + note=" · ⏹ interrupted" if interrupted else "" + print(f"\r {C.dgray}└─ ~{len(''.join(reply))//4} tok · {el:.0f}s{note}{C.r}\n") + print(f" {C.dim}goodbye — the engine keeps running for the next chat 🐦{C.r}") + +def kv_resume_notice(model_dir): + """SERVE mode silently resumes .coli_kv from disk (glm.c kv_disk_load): a chat + started today continues a conversation from days ago, with `first=0` so the + turn is appended WITHOUT the [gMASK] prefix. The engine does announce it + on stderr — but nothing here ever shows that: the drain thread's + p.stderr.read() blocks until EOF, so on a healthy start errlog is still empty + when the status lines are printed. The warning only appeared once the engine + DIED, which is exactly when it no longer mattered. + + Measured cost of the silence: a chat inherited 670 tokens of an old Italian + session ("il mio numero preferito e 7, ricordalo!"). Every later reply came + back in Italian, and "explain fibonacci in short" was answered about the + number 7 — the model was being coherent with a context nobody could see, and + it read as a quantization bug for a day. + + So say it here, in Python, from the file itself: no pipe, no thread, no + Windows deadlock risk (see the stderr comment below).""" + p=os.path.join(model_dir, ".coli_kv") + try: + with open(p,"rb") as f: + if f.read(8)!=b"COLIKV1\0": return + h=struct.unpack("<8i", f.read(32)) + n=h[6] + if n<1: return + age=time.time()-os.path.getmtime(p) + when=f"{age/86400:.0f}d ago" if age>86400 else f"{age/3600:.0f}h ago" if age>3600 else "just now" + print(f" {C.yel}↺ resuming a saved conversation: {n} tokens, last written {when}{C.r}") + print(f" {C.dgray} it steers tone, language and topic. :reset clears it · " + f"KVSAVE=0 disables saving · delete {p} to start clean{C.r}") + except (OSError, struct.error): pass + def cmd_chat(a): + # ATTACH: a running `coli serve` beats a private engine every time — the load + # (34-136 s) and the cache warmth survive between sessions. Explicit --attach + # wins; otherwise probe localhost quietly and use it if it's there. --no-attach + # forces the old behaviour. The probe costs ~1 ms when nothing is listening. + if not getattr(a,"no_attach",False): + base=getattr(a,"attach",None) or "http://127.0.0.1:8000" + mid=server_probe(base, getattr(a,"api_key",None)) + if mid: + banner(f"chat · {mid} · attached") + chat_attached(a, base, mid); return + if getattr(a,"attach",None): + sys.exit(f"--attach: no coli serve answering at {base} (start one with: coli serve --model )") need_model(a.model) banner(f"chat · {os.path.basename(a.model)} · ram {a.ram or '-'}GB · topp {a.topp or 'off'}") + kv_resume_notice(a.model) errlog=tempfile.NamedTemporaryFile(mode="w+", suffix=".log", delete=False) e=env_for(a); e["SERVE"]="1" # stderr -> PIPE, NOT stderr=errlog (file). On Windows/MinGW, pointing the @@ -619,12 +744,65 @@ def cmd_chat(a): except Exception: pass print(f" {C.teal}goodbye{C.r} {C.dim}— the hummingbird returns to its nest{C.r} 🐦\n") +def serve_pidfile(port): return os.path.join(tempfile.gettempdir(), f"coli-serve-{port}.pid") + def cmd_serve(a): need_model(a.model) + # pidfile: cosi' `coli stop` spegne tutto con un comando, senza pkill a mano. + # EN: pidfile so `coli stop` can shut everything down without manual pkill. + try: + with open(serve_pidfile(a.port),"w") as f: f.write(f"{os.getpid()} {a.model}\n") + except OSError: pass from openai_server import serve - serve(a.model, a.host, a.port, a.model_id, a.api_key, - a.cap,a.ngen,GLM,env_for(a),a.cors_origin, - a.max_queue,a.queue_timeout,a.kv_slots) + try: + serve(a.model, a.host, a.port, a.model_id, a.api_key, + a.cap,a.ngen,GLM,env_for(a),a.cors_origin, + a.max_queue,a.queue_timeout,a.kv_slots) + finally: + try: os.unlink(serve_pidfile(a.port)) + except OSError: pass + +def cmd_stop(a): + """Shut down a running `coli serve` AND its engine — one command, no pkill. + The engine re-execs itself for OMP tuning, so its process is named `exe`, + not `glm`: every `pkill -x glm` in history silently killed nothing (that is + how two 17+5 GB ghost engines OOM'd this box on 2026-07-16). This finds the + real processes: the pidfile first, then /proc by cmdline/environ — only + processes that are demonstrably ours (SERVE=1 + our SNAP, or `coli serve` + in the command line).""" + banner("stop") + targets=[] # (pid, descrizione) + pf=serve_pidfile(a.port) + try: + pid=int(open(pf).read().split()[0]) + os.kill(pid,0); targets.append((pid,f"coli serve (pidfile, port {a.port})")) + except (OSError,ValueError,IndexError): pass + for pd in os.listdir("/proc"): + if not pd.isdigit(): continue + pid=int(pd) + try: + cmd=open(f"/proc/{pd}/cmdline","rb").read().replace(b"\0",b" ").decode("utf-8","replace") + if "coli" in cmd and " serve" in cmd and pid!=os.getpid(): + if not any(p==pid for p,_ in targets): targets.append((pid,"coli serve (cmdline)")) + comm=open(f"/proc/{pd}/comm").read().strip() + if comm in ("glm","exe","olmoe"): + env=open(f"/proc/{pd}/environ","rb").read().replace(b"\0",b"\n").decode("utf-8","replace") + if "SERVE=1" in env: targets.append((pid,f"engine `{comm}` (SERVE=1)")) + except (OSError,PermissionError): continue + if not targets: + print(f" nothing running — no serve on port {a.port}, no SERVE engines"); return + for pid,desc in targets: print(f" {'would stop' if a.dry_run else 'stopping'} {pid}: {desc}") + if a.dry_run: return + for pid,_ in targets: + try: os.kill(pid, signal.SIGTERM) + except OSError: pass + time.sleep(2.0) + for pid,_ in targets: + try: os.kill(pid, signal.SIGKILL); print(f" {pid}: forced (SIGKILL)") + except OSError: pass # gia' morto: bene + try: os.unlink(pf) + except OSError: pass + print(f" {C.grn}✓ stopped{C.r} — RAM released") def cmd_web(a): """serve + open the dashboard in the browser once the API answers.""" @@ -719,7 +897,14 @@ def main(): pd=sub.add_parser("doctor",parents=[common]) pd.add_argument("--json",action="store_true",help="emit a versioned JSON report") pr=sub.add_parser("run", parents=[common]); pr.add_argument("prompt", nargs="*") - sub.add_parser("chat", parents=[common]) + pc=sub.add_parser("chat", parents=[common]) + pc.add_argument("--attach", nargs="?", const="http://127.0.0.1:8000", default=None, + help="chat against a running `coli serve` instead of spawning an engine " + "(keeps the model loaded and the expert cache warm across chat sessions). " + "Bare --attach probes localhost:8000.") + pc.add_argument("--no-attach", action="store_true", + help="never auto-attach, always spawn a private engine") + pc.add_argument("--api-key", default=os.environ.get("COLI_API_KEY")) ps=sub.add_parser("serve", parents=[common]) ps.add_argument("--host",default="127.0.0.1"); ps.add_argument("--port",type=int,default=8000) ps.add_argument("--model-id",default=os.environ.get("COLI_MODEL_ID","glm-5.2-colibri")) @@ -728,6 +913,8 @@ def main(): ps.add_argument("--max-queue",type=int,default=int(os.environ.get("COLI_MAX_QUEUE","8"))) ps.add_argument("--queue-timeout",type=float,default=float(os.environ.get("COLI_QUEUE_TIMEOUT","300"))) ps.add_argument("--kv-slots",type=int,default=int(os.environ.get("COLI_KV_SLOTS","1"))) + pst=sub.add_parser("stop", parents=[common], help="shut down a running coli serve and its engine") + pst.add_argument("--port",type=int,default=8000); pst.add_argument("--dry-run",action="store_true") pw=sub.add_parser("web", parents=[common], help="serve + open the dashboard in a browser") for arg,kw in (("--host",dict(default="127.0.0.1")),("--port",dict(type=int,default=8000)), ("--model-id",dict(default=os.environ.get("COLI_MODEL_ID","glm-5.2-colibri"))), @@ -751,7 +938,7 @@ def main(): pc.add_argument("--no-mtp",action="store_true",help="skip the MTP head (no speculative drafts)") a=ap.parse_args() handler={"build":cmd_build,"info":cmd_info,"plan":cmd_plan,"doctor":cmd_doctor, - "run":cmd_run,"chat":cmd_chat,"serve":cmd_serve,"bench":cmd_bench, + "run":cmd_run,"chat":cmd_chat,"serve":cmd_serve,"stop":cmd_stop,"bench":cmd_bench, "convert":cmd_convert,"web":cmd_web}.get(a.cmd) if handler: sys.exit(handler(a) or 0) banner(); print(__doc__) diff --git a/c/compat.h b/c/compat.h index 82de8b6..ce21df2 100644 --- a/c/compat.h +++ b/c/compat.h @@ -143,6 +143,12 @@ static inline int compat_fadvise(int fd, off_t off, off_t len, int advice){ * Thread-safe (no shared seek position). Gestisce offset >4 GB e chunking * per letture >2 GB (anche se i tensori individuali sono nell'ordine dei * MB-centinaia di MB, il wrapper e' robusto per ogni taglia). */ +/* Ultimo GetLastError() di una ReadFile fallita, per thread: il chiamante + * (pread_full in glm.c) lo stampa accanto a strerror. Senza questo, OGNI + * fallimento Windows collassa in "EIO -> Input/output error" e la diagnosi + * dal campo diventa un tirare a indovinare (#307: tre giri di ipotesi tra + * tre persone perche' il codice vero non compariva da nessuna parte). */ +static __thread DWORD compat_pread_lasterr __attribute__((unused)); static inline ssize_t compat_pread(int fd, void *buf, size_t n, off_t off){ intptr_t osfh = _get_osfhandle(fd); if(osfh == -1 || osfh == -2){ errno = EBADF; return -1; } @@ -158,6 +164,7 @@ static inline ssize_t compat_pread(int fd, void *buf, size_t n, off_t off){ if(!ReadFile(h, (char*)buf + total, chunk32, &rd, &ov)){ DWORD err = GetLastError(); if(err == ERROR_HANDLE_EOF) break; /* past EOF → return bytes read (0 if none, matching POSIX pread) */ + compat_pread_lasterr = err; /* preserva il codice VERO per il report (#307) */ if(err == ERROR_INVALID_HANDLE || err == ERROR_INVALID_FUNCTION) errno = EBADF; else errno = EIO; return -1; @@ -237,7 +244,9 @@ static inline int compat_rename(const char *old, const char *new){ /* --- rss_gb: getrusage -> GetProcessMemoryInfo --- * ru_maxrss in KB (come Linux): rss_gb() divide per 1e6 → GB corretti. */ #include -#pragma comment(lib, "psapi.lib") +#ifdef _MSC_VER +#pragma comment(lib, "psapi.lib") /* MSVC: link psapi; MinGW/GCC uses -lpsapi */ +#endif struct rusage { long ru_maxrss; }; #define RUSAGE_SELF 0 static inline int getrusage(int who, struct rusage *r){ diff --git a/c/glm.c b/c/glm.c index 645be12..92a4d78 100644 --- a/c/glm.c +++ b/c/glm.c @@ -40,6 +40,9 @@ #include /* fstat per mmap degli shard (COLI_MMAP) */ #include /* SIGINT = stop morbido del turno in serve mode */ #endif +#ifdef __linux__ +#include /* statfs: real fs-type check for the 9p warning (below) */ +#endif #if defined(_WIN32) && (defined(__x86_64__) || defined(__i386__)) #include /* hwinfo_emit: CPU brand string senza /proc */ #endif @@ -1207,7 +1210,9 @@ static int g_disk_split=0; /* DISK_SPLIT=1: contatori che spezzano i DISK LOAD ( * 10x (#82) — hence per-region mbind here and nothing else. Raw syscall, no libnuma * dependency; MPOL_MF_MOVE migrates pages of reused heap chunks too. Linux-only, * silent no-op elsewhere or on single-node hosts. */ -static int g_numa_nodes=0; +#ifdef __linux__ +static int g_numa_nodes=0; /* only touched under __linux__; off-Linux NUMA is a no-op */ +#endif static void numa_slab_bind(void *p, size_t n){ #ifdef __linux__ if(g_numa_nodes<2 || !p || !n) return; @@ -1732,8 +1737,14 @@ static int pread_full(int fd, void *buf, int64_t n, int64_t off, const char *tag while(goty?-1:0; } +/* PARTIAL SELECT (quickselect, Hoare partition, DESCending). After this call the k + * LARGEST elements of a[0..n) are in a[0..k) in unspecified order; the (k+1)-th and + * beyond are untouched-or-smaller. O(n) average, O(n^2) pathological (mitigated by + * median-of-three below) — and unlike a full qsort it never orders more than needed. + * + * Why this exists (#356): the DSA top-keep in attention_rows previously full-qsorted + * all nk context scores (O(nk log nk)) per layer per token just to read ONE value -- + * the keep-th largest (the threshold). quickselect finds that pivot in O(nk) average, + * and the position-order scans that build dst[] are unchanged, so the kept set is + * bit-identical. Mirrors the sampling-side fix in #335 (heap partial-select there). + * + * NOT a stable partition: callers must derive the threshold and then re-scan the + * ORIGINAL array (the DSA code does exactly this) rather than reading a[0..k). */ +static void partial_select_desc(float *a, int n, int k){ + if(k<=0) return; + if(k>=n) return; /* nothing to partition: all kept */ + int lo=0, hi=n-1; + while(lo>1); + if(a[mid]>a[lo]){ float t=a[lo]; a[lo]=a[mid]; a[mid]=t; } + if(a[hi]>a[lo]){ float t=a[lo]; a[lo]=a[hi]; a[hi]=t; } + if(a[mid]>a[hi]){ float t=a[hi]; a[hi]=a[mid]; a[mid]=t; } + float piv=a[hi]; + int i=lo, j=hi; + for(;;){ + while(a[i]>piv) i++; /* desc: large values go left */ + while(j>lo && a[j]=j) break; + float t=a[i]; a[i]=a[j]; a[j]=t; i++; if(i>j) break; j--; + } + /* partition point: a[lo..i) are all >= piv, a[i..hi] are all <= piv */ + if(k<=i-1) hi=i-1; /* the k-th largest is in the left partition */ + else lo=i; /* it's in the right partition */ + } +} + /* attenzione MLA con KV-cache compressa, su token nuovi x[S,hidden], pos_base = pos del primo */ /* kvs/pos describe a ragged decode batch: each row may belong to a different * sequence. NULL keeps the original contiguous, currently-bound KV path. */ @@ -2586,10 +2634,14 @@ static void attention_rows(Model *m, Layer *l, int layer, float *x, int S, int p } isc[t]=a*wsc; } - /* top-keep: soglia via qsort desc, poi scan in ordine di posizione */ + /* top-keep: threshold via PARTIAL SELECT (#356), poi scan in ordine di posizione. + * Era un qsort completo su nk (O(nk log nk)); quickselect estrae solo il + * keep-esimo valore piu' grande in O(nk) medio. La soglia (= min del blocco + * dei keep maggiori) e' identica a tmp[keep-1] del vecchio qsort, quindi i + * due scan qui sotto costruiscono dst[] bit-identical. */ float *tmp=falloc(nk); memcpy(tmp,isc,nk*sizeof(float)); - qsort(tmp,nk,sizeof(float),cmp_fdesc); - float thr=tmp[keep-1]; + partial_select_desc(tmp,nk,keep); + float thr=tmp[0]; for(int t=1;tdsa_sel+(int64_t)s*dtopk, nd=0; for(int t=0;tthr) dst[nd++]=t; for(int t=0;t>7; g_rng^=g_rng<<17; return (double)(g_rng>>11)*(1.0/9007199254740992.0); } static float *g_pbuf=NULL; static int *g_pidx=NULL; /* buffer riusati (decode single-thread) */ -static int cmp_pdesc(const void *a,const void *b){ - float pa=g_pbuf[*(const int*)a], pb=g_pbuf[*(const int*)b]; - return papb ? -1 : 0; } -/* costruisce in g_pbuf la distribuzione target: softmax(lo/temp) troncata a top-p g_nuc */ +/* sift-down su max-heap in h[0..n), chiave = g_pbuf[h[i]] (#335: partial top-p select). + * Versione "a buco": porta il valore di radice e lo deposita solo alla fine, cosi' + * heapify e' O(V) e ogni pop e' O(log n) senza qsort sull'intero vocabolario. */ +static void topp_siftdown(int *h, int n, int i){ + int iv=h[i]; float kv=g_pbuf[iv]; + for(;;){ int l=2*i+1; + if(l>=n) break; /* foglia */ + int b=l; if(l+1g_pbuf[h[l]]) b=l+1; /* figlio maggiore */ + if(g_pbuf[h[b]]<=kv) break; /* nessun figlio supera la radice -> ferma */ + h[i]=h[b]; i=b; } + h[i]=iv; +} +/* costruisce in g_pbuf la distribuzione target: softmax(lo/temp) troncata a top-p g_nuc. + * Invariante per dist_sample: g_pbuf resta INDICIZZATO per token-id (mai riordinato); + * la coda troncata va AZZERATA in g_pbuf (dist_sample la legge direttamente per id). */ static void dist_build(const float *lo, int V){ if(!g_pbuf){ g_pbuf=falloc(V); g_pidx=malloc(V*sizeof(int)); } float mx=lo[0]; for(int i=1;imx) mx=lo[i]; @@ -4198,12 +4261,19 @@ static void dist_build(const float *lo, int V){ for(int i=0;i0 && g_nuc<1.f){ for(int i=0;i=g_nuc){ keep=i+1; break; } } - double s2=0; for(int i=keep;i=0;i--) topp_siftdown(g_pidx,V,i); /* heapify O(V) */ + /* pop verso la coda: i vincitori (testa top-p) cadono in g_pidx[out..V-1] in ordine + * DECRESCENTE, come il vecchio qsort, quindi s2 accumula nello stesso ordine -> + * head bit-identical sui casi senza pareggi (i pareggi erano gia' non specificati + * sotto il qsort instabile e restano tali). Il prefisso g_pidx[0..out-1) e' la coda. */ + double s2=0, cum=0; int out=V; + do{ int root=g_pidx[0]; /* massimo corrente */ + g_pidx[0]=g_pidx[--out]; g_pidx[out]=root; /* sposta il max in coda */ + s2+=g_pbuf[root]; cum+=g_pbuf[root]; + if(out>0) topp_siftdown(g_pidx,out,0); + } while(cum0); + for(int i=0;i=0 -> quel token e' escluso (rinormalizzando al volo) */ @@ -6000,6 +6070,14 @@ int main(int argc, char **argv){ !getenv("COLI_CUDA") && !getenv("COLI_METAL")){ setenv("OMP_WAIT_POLICY","active",0); /* keep the team hot across the tiny per-expert matmul regions */ setenv("GOMP_SPINCOUNT","200000",0); /* spin briefly, then yield so long disk waits don't burn a core */ + /* LLVM libomp (clang builds: FreeBSD cc, macOS, some Linux setups) does not + * read GOMP_*: with OMP_WAIT_POLICY=active it sets KMP_BLOCKTIME=infinite, + * so the idle team SPINS FOREVER once generation ends — a serve-mode engine + * parked on stdin burns ~100% x nthreads (#341, measured 3000% on FreeBSD). + * 200 ms of blocktime keeps the team hot across back-to-back expert matmuls + * and lets it sleep at the prompt. libgomp ignores KMP_*; overwrite=0 keeps + * the user's own setting authoritative. */ + setenv("KMP_BLOCKTIME","200",0); setenv("OMP_PROC_BIND","close",0); /* pack the team onto adjacent cores for cache locality */ setenv("OMP_DYNAMIC","FALSE",0); /* fixed team size: no per-region thread-count churn */ setenv("COLI_OMP_TUNED","1",1); @@ -6230,9 +6308,16 @@ int main(int argc, char **argv){ m.has_mtp?"ACTIVE":"absent", g_draft); /* anche su stderr: e' il canale che le UI (coli) mostrano all'utente */ fprintf(stderr,"[MTP] %s (draft=%d)\n", m.has_mtp?"active: native speculative decoding":"absent", g_draft); - if(!strncmp(snap,"/mnt/",5)) - fprintf(stderr,"WARNING: the model is on %s (slow 9p/Windows filesystem; fadvise is ineffective).\n" - " Keep it on ext4 (for example, /home/...) for memory efficiency and speed.\n", snap); +#ifdef __linux__ + { /* Only warn for a GENUINE 9p mount (WSL Windows drives, magic 0x01021997), where + * fadvise is a no-op. The old check was `snap` starting with "/mnt/", which + * false-positives on native-Linux ZFS/ext4/xfs/NFS mounts that also live under /mnt. */ + struct statfs sfb; + if(statfs(snap,&sfb)==0 && (unsigned long)sfb.f_type==0x01021997UL) + fprintf(stderr,"WARNING: the model is on %s (9p/Windows filesystem; fadvise is ineffective).\n" + " Keep it on a native Linux fs (ext4/xfs/zfs) for memory efficiency and speed.\n", snap); + } +#endif /* HOT-STORE: PIN= [PIN_GB=g] -> top expert per frequenza fissi in RAM. * Va PRIMA di cap_for_ram: i pinnati contano nel residente. */ if(getenv("PIN")){ diff --git a/c/olmoe.c b/c/olmoe.c index 5923bde..cda5913 100644 --- a/c/olmoe.c +++ b/c/olmoe.c @@ -400,6 +400,37 @@ static void generate(Model *m, const int *prompt, int np, int n_new, int *out) { } } +/* teacher-forced NLL of full_ids[np..nfull): feed the REFERENCE token at each step + * (never the argmax), accumulate -log softmax(logits)[next_ref]. A loss meter for + * throughput experiments: same engine path as decode, so hit rate/speed stay + * comparable, but quality is measured as perplexity instead of exact-match. + * Cross-checked vs HF transformers bf16 on identical token ids: engine (int8 + * experts) 12.11 ppl vs reference 12.25 (#108). Enabled by PPL=1. */ +static int tf_nll(Model *m, const int *full, int nfull, int np, double *nll_out) { + Cfg *c = &m->c; + m->max_t = nfull; + m->K = calloc(c->n_layers, sizeof(float*)); m->V = calloc(c->n_layers, sizeof(float*)); + for (int i = 0; i < c->n_layers; i++) { + m->K[i] = falloc((int64_t)c->n_heads * m->max_t * c->head_dim); + m->V[i] = falloc((int64_t)c->n_heads * m->max_t * c->head_dim); + } + double nll = 0; int scored = 0; + float *logit = step(m, full, np, 0); /* prefill on the prompt */ + for (int i = np; i < nfull; i++) { + /* log softmax(logit)[full[i]] without materializing the softmax */ + float mx = logit[0]; for (int v = 1; v < c->vocab; v++) if (logit[v] > mx) mx = logit[v]; + double Z = 0; for (int v = 0; v < c->vocab; v++) Z += exp((double)logit[v] - mx); + nll += -((double)logit[full[i]] - mx - log(Z)); + scored++; + free(logit); logit = NULL; + if (i == nfull - 1) break; + logit = step(m, &full[i], 1, i); /* teacher forcing */ + } + if (logit) free(logit); + *nll_out = nll / scored; + return scored; +} + /* ---------- lettura ref.json ---------- */ static int *read_int_array(jval *o, const char *key, int *n_out) { jval *a = json_get(o, key); @@ -430,6 +461,19 @@ int main(int argc, char **argv) { Model m; model_init(&m, snap, cap, bits); printf("resident weights loaded in %.1fs | RSS after load: %.2f GB\n", m.dense_load_s, rss_gb()); + if (getenv("PPL") && atoi(getenv("PPL")) == 1) { /* loss-meter mode: teacher-forced NLL */ + double nll; double t = now_s(); + int scored = tf_nll(&m, full, nfull, np, &nll); + double dt = now_s() - t; + double tot = m.hits + m.miss; + printf("TF-NLL: %.4f nats/token over %d tokens | ppl = %.2f\n", nll, scored, exp(nll)); + printf("Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0, + (unsigned long long)m.hits, (unsigned long long)m.miss); + printf("Speed: %.2f tok/s (%.1fs for %d tokens) | PEAK RSS: %.2f GB\n", scored/dt, dt, scored, rss_gb()); + free(buf); free(arena); + return 0; + } + int *out = malloc((np + n_new) * sizeof(int)); double t = now_s(); generate(&m, prompt, np, n_new, out); diff --git a/c/st.h b/c/st.h index 6b4a710..efa6e5a 100644 --- a/c/st.h +++ b/c/st.h @@ -12,6 +12,7 @@ #include #include #include +#include #include #include #include @@ -104,6 +105,38 @@ static int st_direct_fd(shards *S, int fd) { } /* indicizza tutti i model-*.safetensors in snap_dir */ +/* pread completo: chunk-loop (una singola pread si ferma a ~2^31 byte su Linux + * — i tensori bf16 grandi la superano), riprova su EINTR e riporta un errore + * ONESTO: perror stampava "Success" su una short-read (errno resta 0), lo + * stesso sintomo corretto in glm.c per #236. ST_PREAD_CHUNK e' sovrascrivibile + * per i test. EN: full pread — chunk loop (one pread caps at ~2^31 bytes and + * big bf16 tensors exceed it), EINTR retry, honest short-read errors. + * Exits on failure, like every st.h reader. */ +#ifndef ST_PREAD_CHUNK +#define ST_PREAD_CHUNK (1u << 30) +#endif +static void st_pread_full(int fd, void *buf, int64_t n, int64_t off, const char *tag) { + char *p = (char *)buf; + int64_t got = 0; + while (got < n) { + int64_t want = n - got; + if (want > (int64_t)ST_PREAD_CHUNK) want = ST_PREAD_CHUNK; + ssize_t r = pread(fd, p + got, (size_t)want, off + got); + if (r < 0) { + if (errno == EINTR) continue; + fprintf(stderr, "%s: %s (off %lld, %lld/%lld bytes)\n", tag, strerror(errno), + (long long)off, (long long)got, (long long)n); + exit(1); + } + if (r == 0) { + fprintf(stderr, "%s: short read at EOF (off %lld, %lld/%lld bytes) — truncated file?\n", + tag, (long long)off, (long long)got, (long long)n); + exit(1); + } + got += r; + } +} + static void st_init(shards *S, const char *snap_dir) { memset(S, 0, sizeof(*S)); S->cap = 4096; S->t = calloc(S->cap, sizeof(st_tensor)); @@ -128,7 +161,7 @@ static void st_init(shards *S, const char *snap_dir) { if (fstat(fd, &sst) != 0) { perror("fstat shard"); exit(1); } int64_t fsz = (int64_t)sst.st_size; uint64_t hlen; - if (pread(fd, &hlen, 8, 0) != 8) { perror("pread hlen"); exit(1); } + st_pread_full(fd, &hlen, 8, 0, "pread hlen"); /* file malevolo/troncato: hlen deve stare nel file dopo gli 8 byte di * prefisso e sotto il tetto. Senza questo bound hlen+1 puo' andare in * overflow (malloc(0) e poi hdr[hlen]=0 fuori limiti) o forzare una @@ -138,7 +171,7 @@ static void st_init(shards *S, const char *snap_dir) { files[fi], (unsigned long long)hlen, (long long)fsz); exit(1); } char *hdr = malloc(hlen + 1); if (!hdr) { perror("malloc safetensors header"); exit(1); } - if (pread(fd, hdr, hlen, 8) != (ssize_t)hlen) { perror("pread hdr"); exit(1); } + st_pread_full(fd, hdr, (int64_t)hlen, 8, "pread hdr"); hdr[hlen] = 0; int64_t data_start = 8 + (int64_t)hlen; char *arena = NULL; @@ -218,7 +251,7 @@ static int64_t st_read_f32(shards *S, const char *name, float *out, int drop) { if (!t) { fprintf(stderr, "missing tensor: %s\n", name); exit(1); } void *raw = malloc(t->nbytes); if (!raw) { fprintf(stderr, "malloc %lld bytes for tensor %s failed\n", (long long)t->nbytes, name); exit(1); } - if (pread(t->fd, raw, t->nbytes, t->off) != t->nbytes) { perror("pread data"); exit(1); } + st_pread_full(t->fd, raw, t->nbytes, t->off, "pread data"); if (t->dtype == 2) { memcpy(out, raw, t->nbytes); } else if (t->dtype == 0) { @@ -243,7 +276,7 @@ static int64_t st_nbytes(shards *S, const char *name) { static void st_read_raw(shards *S, const char *name, void *out, int drop) { st_tensor *t = st_find(S, name); if (!t) { fprintf(stderr, "missing tensor: %s\n", name); exit(1); } - if (pread(t->fd, out, t->nbytes, t->off) != t->nbytes) { perror("pread raw"); exit(1); } + st_pread_full(t->fd, out, t->nbytes, t->off, "pread raw"); if (drop) posix_fadvise(t->fd, t->off, t->nbytes, POSIX_FADV_DONTNEED); } @@ -256,7 +289,7 @@ static void st_read_slice_f32(shards *S, const char *name, int64_t elem_off, int int esz = (t->dtype == 2) ? 4 : 2; int64_t boff = t->off + elem_off * esz, nb = n_elems * esz; void *raw = malloc(nb); - if (pread(t->fd, raw, nb, boff) != nb) { perror("pread slice"); exit(1); } + st_pread_full(t->fd, raw, nb, boff, "pread slice"); if (t->dtype == 2) memcpy(out, raw, nb); else if (t->dtype == 0) { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = bf16_to_f32(p[i]); } else { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = f16_to_f32(p[i]); } diff --git a/c/tests/bench_dsa_select.c b/c/tests/bench_dsa_select.c new file mode 100644 index 0000000..1fd3e6b --- /dev/null +++ b/c/tests/bench_dsa_select.c @@ -0,0 +1,132 @@ +/* Microbenchmark: old (full-qsort) vs new (quickselect partial-select) DSA top-keep. + * + * This is NOT a unit test -- test_dsa_select.c proves correctness. This measures the + * headline claim of #356: that replacing the O(nk log nk) qsort over all nk context + * scores with an O(nk) partial_select_desc is materially faster per call, which is + * the win the issue was opened for -- and that the win GROWS with context length + * (because quickselect is linear average, qsort is n-log-n). + * + * It re-implements the OLD top-keep inline (qsort + threshold + scans) on a private + * buffer so the A/B runs in one process, same inputs, same warm caches -- a controlled + * comparison. It calls the REAL (new) partial_select_desc via the include-glm.c + * pattern, replicating the production threshold derivation + scans. + * + * Methodology (chosen to be honest, not to flatter the change): + * - keep = 2048 (the real GLM-5.2 index_topk), nk swept across context lengths from + * the 2049 activation boundary up to 65536 (a long conversation). + * - Three score shapes: (a) realistic peaked -- a few hot keys, long tail, the shape + * real DSA attention scores take; (b) uniform random -- no structure; (c) a plateau + * of ties, to exercise the boundary-membership path. + * - Each (shape, nk) is timed over N_REPEAT=2000 iterations, with the scores frozen + * so both algorithms do IDENTICAL work. We report median ns/call and the new/old + * ratio. A warmup pass primes caches before timing. + * + * Run: make tests/bench_dsa_select && ./tests/bench_dsa_select (not in TEST_BINS) + */ +#define main coli_glm_main_unused +#include "../glm.c" +#undef main + +#include +#include + +/* ---- the OLD algorithm, verbatim from dev before #356, on a private buffer ---- */ +static int cmp_pdesc_old(const void *a, const void *b){ + float x=*(const float*)a, y=*(const float*)b; return xy?-1:0; } +static void keep_old(const float *isc, int nk, int keep, int *dst, int *nd_out){ + float *tmp=malloc((size_t)nk*sizeof(float)); + memcpy(tmp,isc,(size_t)nk*sizeof(float)); + qsort(tmp,(size_t)nk,sizeof(float),cmp_pdesc_old); + float thr=tmp[keep-1]; int nd=0; + for(int t=0;tthr) dst[nd++]=t; + for(int t=0;t=0 && ts[b]>k){ ts[b+1]=ts[b]; b--; } ts[b+1]=k; } + free(dst); + return ts[N_REPEAT/2]; +} + +/* the NEW algorithm calls the real partial_select_desc + replicates the production + * threshold derivation and position scans. */ +static void keep_new(const float *isc, int nk, int keep, int *dst, int *nd_out){ + float *tmp=malloc((size_t)nk*sizeof(float)); + memcpy(tmp,isc,(size_t)nk*sizeof(float)); + partial_select_desc(tmp,nk,keep); + float thr=tmp[0]; for(int t=1;tthr) dst[nd++]=t; + for(int t=0;t> 17; brng ^= brng << 5; + return (double)(brng >> 8) * (1.0 / 16777216.0); } +static void fill_realistic(float *isc, int nk){ /* few hot, long distinct tail */ + for(int i=0;i boundary path */ + for(int i=0;i +#include + +/* ---- the OLD algorithm, verbatim from dev before #354, on a private buffer ---- */ +static float *s_pbuf; static int *s_pidx; static double *s_ref; +static int cmp_pdesc_old(const void *a, const void *b){ + double pa = s_ref[*(const int*)a], pb = s_ref[*(const int*)b]; + return pa < pb ? 1 : pa > pb ? -1 : 0; } +static void dist_build_old(const float *lo, int V, double temp, double nuc){ + double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i]; + double s = 0, invt = 1.0 / (temp > 1e-4 ? temp : 1e-4); + for (int i = 0; i < V; i++){ s_ref[i] = exp((lo[i]-mx)*invt); s += s_ref[i]; } + for (int i = 0; i < V; i++) s_ref[i] /= s; + if (nuc > 0 && nuc < 1.0){ + for (int i = 0; i < V; i++) s_pidx[i] = i; + qsort(s_pidx, V, sizeof(int), cmp_pdesc_old); + double cum = 0; int keep = V; + for (int i = 0; i < V; i++){ cum += s_ref[s_pidx[i]]; if (cum >= nuc){ keep = i+1; break; } } + double s2 = 0; + for (int i = keep; i < V; i++) s_ref[s_pidx[i]] = 0; + for (int i = 0; i < keep; i++) s2 += s_ref[s_pidx[i]]; + for (int i = 0; i < keep; i++) s_ref[s_pidx[i]] /= s2; + } + (void)s_pbuf; +} + +/* ---- timing: median of N_REPEAT runs in ns/call, sorted ascending ---- */ +#define N_REPEAT 2000 +static double bench_ns(void (*fn)(const float*,int,double,double), + const float *lo, int V, double temp, double nuc){ + static double ts[N_REPEAT]; + for (int r = 0; r < N_REPEAT; r++){ + double t0 = now_s(); + fn(lo, V, temp, nuc); + ts[r] = (now_s() - t0) * 1e9; + } + /* insertion sort the N_REPEAT samples (small), take median */ + for (int a = 1; a < N_REPEAT; a++){ double k = ts[a]; int b = a-1; + while (b >= 0 && ts[b] > k){ ts[b+1] = ts[b]; b--; } ts[b+1] = k; } + return ts[N_REPEAT/2]; +} + +/* the NEW algorithm is the real dist_build, but it writes g_pbuf (not a private buf). + * Wrap it so the bench signature matches, and set the globals it reads. */ +static void dist_build_new(const float *lo, int V, double temp, double nuc){ + g_temp = (float)temp; g_nuc = (float)nuc; + dist_build(lo, V); +} + +/* deterministic logit fill for three shapes */ +static uint32_t brng = 0xA5A5A5A5u; +static double brand(void){ brng ^= brng << 13; brng ^= brng >> 17; brng ^= brng << 5; + return (double)(brng >> 8) * (1.0 / 16777216.0); } +static void fill_realistic(float *lo, int V){ /* one hot, exponential tail -- like real logits */ + for (int i = 0; i < V; i++) lo[i] = (float)(-4.0 * brand() - (double)i * 0.0001); + lo[0] = 6.f; lo[V/50] = 4.f; lo[V/200] = 3.f; +} +static void fill_uniform(float *lo, int V){ /* worst case for the heap: max pop count */ + for (int i = 0; i < V; i++) lo[i] = 0.f; +} +static void fill_plateau(float *lo, int V){ /* ties: blocks of equal value */ + for (int i = 0; i < V; i++) lo[i] = (float)(-(double)(i / 50)); +} + +int main(void){ + int V = 151936; + float *lo = malloc((size_t)V * sizeof(float)); + s_ref = malloc((size_t)V * sizeof(double)); + s_pidx = malloc((size_t)V * sizeof(int)); + /* force the new dist_build to allocate g_pbuf/g_pidx at full V once */ + g_temp = 0.7f; g_nuc = 0.9f; dist_build(lo, V); + + double temp = 0.7; + struct { const char *name; void (*fill)(float*,int); } shapes[] = { + { "realistic", fill_realistic }, + { "uniform", fill_uniform }, + { "plateau", fill_plateau }, + }; + double nucs[] = { 0.5, 0.9, 0.95, 0.99 }; + + printf("bench_topp: top-p truncation, old (qsort) vs new (heap) V=%d temp=%.2f\n", V, temp); + printf("%-12s %6s %14s %14s %9s %9s\n", "shape", "nuc", "old ns/call", "new ns/call", "speedup", "keep"); + printf("-----------------------------------------------------------------------------\n"); + + for (size_t sh = 0; sh < sizeof(shapes)/sizeof(shapes[0]); sh++){ + shapes[sh].fill(lo, V); + for (size_t ni = 0; ni < sizeof(nucs)/sizeof(nucs[0]); ni++){ + double nuc = nucs[ni]; + /* warmup both paths so caches/branch predictors are primed */ + for (int w = 0; w < 50; w++){ dist_build_old(lo, V, temp, nuc); dist_build_new(lo, V, temp, nuc); } + double t_old = bench_ns(dist_build_old, lo, V, temp, nuc); + double t_new = bench_ns(dist_build_new, lo, V, temp, nuc); + /* keep count = non-zero entries the new path leaves (== old's keep) */ + int keep = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) keep++; + printf("%-12s %6.2f %14.0f %14.0f %8.2fx %9d\n", + shapes[sh].name, nuc, t_old, t_new, t_old / t_new, keep); + } + printf("\n"); + } + + printf("bench_topp: done (lower ns is better; speedup = old/new)\n"); + free(lo); free(s_ref); free(s_pidx); + return 0; +} diff --git a/c/tests/test_dsa_select.c b/c/tests/test_dsa_select.c new file mode 100644 index 0000000..99fecb0 --- /dev/null +++ b/c/tests/test_dsa_select.c @@ -0,0 +1,223 @@ +/* DSA top-keep partial-select: the quickselect rewrite (#356) must produce a + * BIT-IDENTICAL kept-position set to the old full-vocab qsort, for every score + * shape the attention indexer can see. + * + * Why this test exists (#356): attention_rows() selects the top-`keep` context + * keys (index_topk=2048 on GLM-5.2) to attend to. It previously did this by + * full-qsorting all `nk` scores (O(nk log nk)) and reading tmp[keep-1] as the + * threshold. It now does a partial_select_desc (quickselect, O(nk) average) and + * takes the threshold as the min of the selected top-keep block. The contract is + * subtle but STRONGER than test_topp's: + * + * The two position-order scans that build dst[] -- + * for t: if isc[t] > thr -> keep (strictly above threshold) + * for t: if isc[t] == thr -> keep (ties, in position order) + * -- are UNCHANGED by the rewrite. So if the threshold value is identical, + * the kept-position set is identical element-by-element (not just as a + * multiset, which is all the unstable sampling heap in #335 could promise). + * + * Strategy: drive the REAL partial_select_desc (via the include-glm.c pattern) + * and replicate the production threshold derivation + scans, then compare the + * resulting dst[] against an INDEPENDENT reference that re-implements the OLD + * algorithm (full qsort + tmp[keep-1] threshold) on a private buffer. The kept + * sets must be element-wise equal on every shape, including tie plateaus where + * the boundary membership is decided by the position scan. + * + * We also directly unit-test partial_select_desc's partition invariant: after + * the call, max(a[keep..n)) <= min(a[0..keep)) -- i.e. the keep largest really + * did land in the prefix. This catches a broken quickselect even before the + * end-to-end comparison. + * + * In-memory only (no scratch files), so it builds clean on the Windows MinGW CI + * job without the unmerged compat shim. */ +#define main coli_glm_main_unused +#include "../glm.c" +#undef main + +#include + +static int g_nfails = 0; + +#define FAIL(fmt, ...) do { \ + fprintf(stderr, " FAIL [%s nk=%d keep=%d shape=%s]: " fmt "\n", \ + label, nk, keep, shape_name, ##__VA_ARGS__); \ + g_nfails++; \ + return; \ +} while (0) + +/* ---- independent reference: the OLD algorithm (full qsort + tmp[keep-1]) ---- */ +/* qsort comparator matching the production cmp_fdesc exactly (desc, unstable). */ +static int cmp_ref_desc(const void *a, const void *b){ + float x=*(const float*)a, y=*(const float*)b; return xy?-1:0; } + +/* Reproduce the OLD glm.c:2589-2596 exactly: copy, qsort desc, threshold = + * tmp[keep-1], then the two position-order scans into dst[]. Returns nd. */ +static int keep_old(const float *isc, int nk, int keep, int *dst){ + float *tmp=malloc((size_t)nk*sizeof(float)); + memcpy(tmp,isc,(size_t)nk*sizeof(float)); + qsort(tmp,(size_t)nk,sizeof(float),cmp_ref_desc); + float thr=tmp[keep-1]; + int nd=0; + for(int t=0;tthr) dst[nd++]=t; + for(int t=0;tthr) dst[nd++]=t; + for(int t=0;t= every element of tmp[keep..n). + * (>=, not >: equal values may sit on either side of the partition boundary, + * which is fine -- the threshold is the MIN of the prefix, and the position + * scan handles ties.) */ + float top_min=INFINITY, tail_max=-INFINITY; + for(int i=0;itail_max) tail_max=tmp[i]; + if(!(top_min >= tail_max)) + FAIL("partition invariant violated: top_min=%.9g < tail_max=%.9g", top_min, tail_max); + free(tmp); +} + +/* ---- end-to-end: old vs new kept-set must be element-wise identical ---- */ +static void check_case(const char *label, int nk, int keep, const char *shape_name, + const float *isc){ + int *da=malloc((size_t)nk*sizeof(int)); + int *db=malloc((size_t)nk*sizeof(int)); + int na=keep_old(isc,nk,keep,da); + int nb=keep_new(isc,nk,keep,db); + + /* 1. both keep exactly `keep` positions (the contract: keep the top-keep by + * count). A count mismatch is a real bug, not a tie artifact. */ + if(na!=keep) FAIL("old kept %d, expected %d (old path is the reference)", na, keep); + if(nb!=keep) FAIL("new kept %d, expected %d", nb, keep); + if(na!=nb) FAIL("keep-count mismatch: old=%d new=%d", na, nb); + + /* 2. element-wise identical dst[]. This is the strong contract: because the + * threshold is derived identically and the position-order scans are byte- + * for-byte the same, the kept SET and its ORDER must match exactly. (This + * is what makes #356 cleaner than #335, which was multiset-only.) */ + int first_diff=-1; + for(int i=0;i=0) + FAIL("kept-set differs at index %d: old dst[%d]=%d new dst[%d]=%d", + first_diff, first_diff, da[first_diff], first_diff, db[first_diff]); + + /* 3. also check the partition invariant directly (catches a subtly broken + * quickselect even if the threshold happened to come out right). */ + check_partition(label,isc,nk,keep,shape_name); + + free(da); free(db); + printf(" ok [nk=%d keep=%d shape=%s]\n", nk, keep, shape_name); +} +#undef FAIL + +/* deterministic xorshift32 RNG (matches the test_i4_grouped.c / test_topp.c convention) */ +static uint32_t rng_state = 0x12345678u; +static uint32_t xr(void){ rng_state ^= rng_state << 13; rng_state ^= rng_state >> 17; + rng_state ^= rng_state << 5; return rng_state; } +static double frand(void){ return (xr() >> 8) * (1.0 / 16777216.0); } /* [0,1) */ + +/* fill scores for a given shape. Shapes stress the threshold boundary and the + * quickselect's median-of-three pivot differently. */ +static void fill_shape(float *isc, int nk, int shape){ + switch(shape){ + case 0: /* uniform random distinct (no ties): the clean contract case */ + for(int i=0;i3) isc[nk/3]=1.f; if(nk>2) isc[nk/2]=0.5f; break; + case 2: /* strictly decreasing geometric (no ties): sorted input -- worst case + * for a naive quickselect; median-of-three must handle it */ + for(int i=0;i boundary membership decided + * entirely by the position scan (exercises the ==thr path) */ + for(int i=0;i degenerate threshold, all kept + * via the ==thr scan; quickselect must not infinite-loop or corrupt */ + for(int i=0;i=n / k==1 / k==n edges. */ + int nks[] = {1, 2, 8, 64, 2049, 4097, 8193}; + int keeps[] = {1, 8, 256, 1024, 2048}; + int n_shapes = 6; + + int cases = 0; + for(size_t ni=0; nin, but skip the plateau/geometric edge if nk<7) */ + fill_shape(isc,nk,shape); + for(size_t ki=0; kink) continue; /* keep<=nk invariant of the production code */ + if(keep<=0) continue; + char label[40]; snprintf(label,sizeof(label),"nk[%zu]/keep[%zu]/shape[%d]",ni,ki,shape); + const char *sn=(const char*[]){"random","peaked","decreasing","increasing","plateau","all-equal"}[shape]; + check_case(label,nk,keep,sn,isc); + cases++; + } + } + free(isc); + } + + /* edge: keep == nk (nothing to partition; both paths keep everything) */ + { + int nk=100, keep=100; float isc[100]; + for(int i=0;i every kept slot is a tie; + * the position scan must pick positions 0..keep-1 deterministically */ + { + int nk=1000, keep=500; float isc[1000]; + for(int i=0;i positions 0..%d]\n",keep,keep-1); cases++; + free(db); + } + + printf("\ntest_dsa_select: %d cases run, %d failure(s)\n", cases, g_nfails); + if(g_nfails){ printf("test_dsa_select: FAIL\n"); return 1; } + printf("test_dsa_select: ok\n"); + return 0; +} diff --git a/c/tests/test_st_pread.c b/c/tests/test_st_pread.c new file mode 100644 index 0000000..c874c08 --- /dev/null +++ b/c/tests/test_st_pread.c @@ -0,0 +1,89 @@ +/* st_pread_full: chunk loop + honest truncation errors. + * Built with -DST_PREAD_CHUNK=7 so a ~100-byte tensor takes many pread calls — + * exercising the loop that production only needs past 2^31 bytes (one pread + * caps there on Linux; big bf16 tensors exceed it). Also forks a child against + * a truncated shard and requires exit(1) with a "short read" message instead + * of the old perror("... : Success"). */ +#define _GNU_SOURCE +#include +#include +#include +#ifndef _WIN32 +#include +#include +#endif + +#include "../st.h" + +#define CHECK(condition) do { \ + if (!(condition)) { \ + fprintf(stderr, "%s:%d: check failed: %s\n", __FILE__, __LINE__, #condition); \ + return 1; \ + } \ +} while (0) + +static void write_snap(const char *dir, int truncate_bytes) { + char path[512]; + snprintf(path, sizeof(path), "%s/model.safetensors", dir); + unsigned char data[96]; + for (int i = 0; i < 96; i++) data[i] = (unsigned char)(i * 7 + 3); + const char *hdr = "{\"t\":{\"dtype\":\"U8\",\"shape\":[96],\"data_offsets\":[0,96]}}"; + uint64_t hlen = strlen(hdr); + FILE *f = fopen(path, "wb"); + fwrite(&hlen, 8, 1, f); + fwrite(hdr, 1, hlen, f); + fwrite(data, 1, (size_t)(96 - truncate_bytes), f); + fclose(f); +} + +int main(void) { + /* relative to the CWD, per test_stops: MinGW .exe files resolve Windows + * paths and "/tmp" is not one */ + char dir[] = "test_st_pread_XXXXXX"; + if (!mkdtemp(dir)) { perror("mkdtemp"); return 1; } + + /* 1) chunk loop: 96-byte tensor read 7 bytes at a time, content exact */ + write_snap(dir, 0); + shards S; st_init(&S, dir); + unsigned char out[96] = {0}; + st_read_raw(&S, "t", out, 0); + for (int i = 0; i < 96; i++) CHECK(out[i] == (unsigned char)(i * 7 + 3)); + +#ifndef _WIN32 + /* 2) shard truncated AFTER st_init (init validates static bounds, so the + * pread path only fires when the file shrinks underneath a live handle): + * child must exit(1) with an honest message, not perror's "Success" */ + char shard[512]; snprintf(shard, sizeof(shard), "%s/model.safetensors", dir); + struct stat sb; CHECK(stat(shard, &sb) == 0); + CHECK(truncate(shard, sb.st_size - 40) == 0); + int pipefd[2]; CHECK(pipe(pipefd) == 0); + pid_t pid = fork(); CHECK(pid >= 0); + if (pid == 0) { + dup2(pipefd[1], 2); close(pipefd[0]); close(pipefd[1]); + unsigned char buf[96]; + st_read_raw(&S, "t", buf, 0); /* inherited handles; must exit(1) inside */ + _exit(42); /* reaching here = bug */ + } + close(pipefd[1]); + char err[512] = {0}; + ssize_t n = read(pipefd[0], err, sizeof(err)-1); (void)n; + close(pipefd[0]); + int status = 0; waitpid(pid, &status, 0); + CHECK(WIFEXITED(status) && WEXITSTATUS(status) == 1); + CHECK(strstr(err, "short read") != NULL); + CHECK(strstr(err, "Success") == NULL); +#else + /* fork/pipe/truncate are POSIX; Windows still runs the chunk-loop check */ + printf("test_st_pread: truncation subtest skipped on Windows\n"); +#endif + + char cmd[600]; +#ifdef _WIN32 + snprintf(cmd, sizeof(cmd), "rmdir /s /q %s", dir); +#else + snprintf(cmd, sizeof(cmd), "rm -rf %s", dir); +#endif + if (system(cmd)) {} + printf("test_st_pread: chunk loop + honest truncation error: ok\n"); + return 0; +} diff --git a/c/tests/test_topp.c b/c/tests/test_topp.c new file mode 100644 index 0000000..9276507 --- /dev/null +++ b/c/tests/test_topp.c @@ -0,0 +1,280 @@ +/* Top-p (nucleus) truncation in dist_build: the partial-select rewrite (#335) must be + * indistinguishable from the old full-vocab qsort for every shape dist_sample can see. + * + * Why this test exists (#335): dist_build() previously qsort-ed the entire 151936-entry + * vocab on every sampled token to find the few-hundred-token head whose cumulative mass + * reaches g_nuc. It now heapifies (O(V)) and pops only the head (k * O(log V)). The win + * is structural; the risk is a silent sampling-distribution change, because the contract + * is subtle: + * + * dist_sample() iterates g_pbuf[0..V-1] BY TOKEN ID and sums probabilities directly. + * So dist_build MUST leave g_pbuf indexed by id (never reordered) AND must zero every + * truncated tail entry -- merely excluding the tail from the head would leave mass on + * it and the sampled distribution would drift with no crash and no error. + * + * Strategy: drive the REAL dist_build (via the test_stops.c include-glm.c pattern) on a + * sweep of distributions and g_nuc values, and compare against an INDEPENDENT reference + * that re-implements the OLD algorithm (full qsort + zero-tail + renorm) in double on a + * private buffer. On shapes with no ties the renormalized head must be BIT-IDENTICAL to + * the reference (the issue's stated invariant: s2 accumulates in the same descending + * order). On tie shapes, where the unstable qsort already left ordering unspecified, we + * check multiset equality instead. Every shape also checks: exact-zero tails, head sums + * to 1.0, and a sane keep-count. + * + * No scratch files: the test runs entirely in memory (no mkdtemp), so it builds clean on + * the Windows MinGW CI job without the unmerged compat shim (#352). */ +#define main coli_glm_main_unused +#include "../glm.c" +#undef main + +#include + +static int g_nfails = 0; + +/* pointer set by ref_build so cmp_ref_desc can read the current reference buffer + * (the qsort comparator gets no user-data argument in C). */ +static const double *g_ref_p = NULL; + +#define FAIL(fmt, ...) do { \ + fprintf(stderr, " FAIL [%s V=%d nuc=%.3f shape=%s]: " fmt "\n", \ + label, V, nuc, shape_name, ##__VA_ARGS__); \ + g_nfails++; \ + return; \ +} while (0) + +/* ---- independent reference: the OLD algorithm, in double, on a private buffer ------- */ +/* Stable qsort by descending probability (ties broken by ascending index, which makes + * the reference deterministic regardless of the production comparator). */ +static int cmp_ref_desc(const void *a, const void *b){ + double pa = ((const double *)g_ref_p)[*(const int*)a]; + double pb = ((const double *)g_ref_p)[*(const int*)b]; + if (pa < pb) return 1; + if (pa > pb) return -1; + /* tie -> lower index first (stable, unlike the production comparator) */ + return *(const int*)a - *(const int*)b; +} + +/* Build the reference distribution into out[0..V-1] (indexed by token id), mirroring the + * old dist_build: softmax(lo/temp) truncated to top-p nuc, tail zeroed, head renormalized. + * Returns the keep-count through *keep_out. */ +static void ref_build(const float *lo, int V, double temp, double nuc, + double *out, int *pidx, int *keep_out){ + double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i]; + double s = 0, invt = 1.0 / (temp > 1e-4 ? temp : 1e-4); + for (int i = 0; i < V; i++){ out[i] = exp((lo[i]-mx)*invt); s += out[i]; } + for (int i = 0; i < V; i++) out[i] /= s; + + if (nuc > 0 && nuc < 1.0){ + for (int i = 0; i < V; i++) pidx[i] = i; + qsort(pidx, V, sizeof(int), cmp_ref_desc); + double cum = 0; int keep = V; + for (int i = 0; i < V; i++){ cum += out[pidx[i]]; if (cum >= nuc){ keep = i+1; break; } } + double s2 = 0; + for (int i = keep; i < V; i++) out[pidx[i]] = 0; + for (int i = 0; i < keep; i++) s2 += out[pidx[i]]; + for (int i = 0; i < keep; i++) out[pidx[i]] /= s2; + *keep_out = keep; + } else { + *keep_out = V; + } +} + +/* count how many production g_pbuf entries are non-zero == the head size */ +static int head_count(int V){ + int n = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) n++; return n; +} + +/* Run one case: load logits into g_pbuf via the real dist_build, compare to reference. + * shape_name is for diagnostics only. */ +static void check_case(const char *label, int V, double nuc, const char *shape_name, + const float *lo){ + /* reference on a private buffer */ + double *ref = malloc((size_t)V * sizeof(double)); + int *ridx = malloc((size_t)V * sizeof(int)); + int ref_keep = 0; + g_ref_p = ref; /* cmp_ref_desc reads this */ + ref_build(lo, V, g_temp, nuc, ref, ridx, &ref_keep); + + /* production: drive the real dist_build (writes the global g_pbuf) */ + g_nuc = (float)nuc; + dist_build(lo, V); + + int got_keep = head_count(V); + + /* 1. keep-count must match the reference exactly. The partial select and the old + * qsort keep the same NUMBER of tokens by construction (same cumulative-mass rule); + * a count divergence is a real bug, not a tie artifact. */ + if (got_keep != ref_keep) + FAIL("keep-count mismatch: got %d, ref %d", got_keep, ref_keep); + + /* 2. Detect ties across the WHOLE pre-truncation distribution, not just the kept set. + * A tie at the head/tail boundary makes which-side-a-token-lands-on interchangeable: + * both algorithms keep the right count but may keep different MEMBERS. So any input + * with a duplicated softmax value needs the relaxed multiset comparison below. We + * detect this on the reference softmax (pre-truncation) by sorting all V values. */ + int has_ties = 0; + { + double *all = malloc((size_t)V * sizeof(double)); + /* reconstruct the pre-truncation softmax the same way ref_build does */ + double mx = lo[0]; for (int i = 1; i < V; i++) if (lo[i] > mx) mx = lo[i]; + double s = 0, invt = 1.0 / (g_temp > 1e-4 ? g_temp : 1e-4); + for (int i = 0; i < V; i++){ all[i] = exp((lo[i]-mx)*invt); s += all[i]; } + for (int i = 0; i < V; i++) all[i] /= s; + for (int a = 1; a < V; a++){ double k = all[a]; int b = a-1; + while (b >= 0 && all[b] > k){ all[b+1] = all[b]; b--; } all[b+1] = k; } + for (int a = 1; a < V; a++) if (all[a] == all[a-1]){ has_ties = 1; break; } + free(all); + } + + if (has_ties){ + /* Multiset equality of the non-zero (head) values. Ties make membership + * interchangeable, so we compare sorted value-multisets, not id-aligned values. + * Tolerance is 1e-6 relative -- the engine uses float arithmetic, the reference + * double, so sub-ULP noise is expected (matches test_i4_grouped.c's convention). */ + double *got = malloc((size_t)ref_keep * sizeof(double)); + int gm = 0; + for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) got[gm++] = (double)g_pbuf[i]; + if (gm != ref_keep) + FAIL("tie-shape head size mismatch: got %d non-zero, ref %d", gm, ref_keep); + for (int a = 1; a < gm; a++){ double k = got[a]; int b = a-1; + while (b >= 0 && got[b] > k){ got[b+1] = got[b]; b--; } got[b+1] = k; } + double *rsort = malloc((size_t)ref_keep * sizeof(double)); + int rm = 0; + for (int i = 0; i < V; i++) if (ref[i] != 0.0) rsort[rm++] = ref[i]; + for (int a = 1; a < rm; a++){ double k = rsort[a]; int b = a-1; + while (b >= 0 && rsort[b] > k){ rsort[b+1] = rsort[b]; b--; } rsort[b+1] = k; } + int mm = 0; double worst = 0; + for (int i = 0; i < gm; i++){ + double d = fabs(got[i] - rsort[i]); + double rel = rsort[i] > 1e-30 ? d / rsort[i] : d; + if (rel > worst) worst = rel; + if (rel > 1e-6) mm++; + } + free(got); free(rsort); + if (mm) FAIL("tie-shape multiset mismatch: %d/%d head values differ beyond 1e-6 rel (worst %.3g)", + mm, ref_keep, worst); + } else { + /* No ties anywhere: membership is forced, so compare id-aligned head values. The + * engine computes in float (g_pbuf /= (float)s2) while the reference uses double, + * so the comparison is relative-tolerance (1e-6), not bit-exact -- the partial + * select and qsort accumulate s2 in the same descending order, so any difference + * is pure float-rounding noise, not an ordering bug. */ + int bad = 0; int first_id = -1; float gv = 0, rv = 0; double worst = 0; + for (int i = 0; i < V; i++){ + if (ref[i] == 0.0) continue; /* tail */ + float want = (float)ref[i]; + double d = fabs((double)g_pbuf[i] - (double)want); + double rel = fabs((double)want) > 1e-30 ? d / fabs((double)want) : d; + if (rel > worst) worst = rel; + if (rel > 1e-6){ + bad++; if (first_id < 0){ first_id = i; gv = g_pbuf[i]; rv = want; } + if (bad > 3) break; + } + } + if (bad) + FAIL("head not within 1e-6 rel of reference: %d entries differ (first id %d: got %.9g want %.9g, worst %.3g)", + bad, first_id, (double)gv, (double)rv, worst); + } + + /* 3. head must renormalize to 1.0 (within float epsilon) */ + double sum = 0; for (int i = 0; i < V; i++) sum += g_pbuf[i]; + if (fabs(sum - 1.0) > 1e-5) + FAIL("head does not sum to 1.0: sum=%.12g (keep=%d)", sum, got_keep); + + free(ref); free(ridx); + printf(" ok [V=%d nuc=%.3f shape=%s keep=%d%s sum=%.10f]\n", + V, nuc, shape_name, got_keep, has_ties ? " (ties)" : "", sum); +} +#undef FAIL + +/* deterministic xorshift32 RNG (matches the test_i4_grouped.c convention) */ +static uint32_t rng_state = 0x12345678u; +static uint32_t xr(void){ rng_state ^= rng_state << 13; rng_state ^= rng_state >> 17; + rng_state ^= rng_state << 5; return rng_state; } +static double frand(void){ return (xr() >> 8) * (1.0 / 16777216.0); } /* [0,1) */ + +/* fill logits for a given shape. Shapes chosen to stress the comparator and the head/tail + * boundary differently. */ +static void fill_shape(float *lo, int V, int shape){ + switch (shape){ + case 0: /* uniform -> every token equal probability -> massive tie plateau */ + for (int i = 0; i < V; i++) lo[i] = 0.f; break; + case 1: /* peaked: one dominant token, rest small and distinct (no ties). + * The fixed hot-spots are clamped to V-1 so small V (incl. V=1) doesn't + * write out of bounds and corrupt heap metadata on the later free(lo). */ + for (int i = 0; i < V; i++) lo[i] = (float)(-1.0 - frand()*4.0); + lo[0] = 3.f; lo[V/3 comparator tie handling */ + for (int i = 0; i < V; i++) lo[i] = (float)(-(double)(i / 7)); /* 7-wide plateaus */ + break; + case 4: /* sharp-tail: a few hot, then a long flat floor (small tie at the floor). + * Hot count is min(12,V) so V<12 (incl. V=1) stays in bounds. */ + for (int i = 0; i < V; i++) lo[i] = -8.f; + { int hot = V<12 ? V : 12; for (int i = 0; i < hot; i++) lo[i] = (float)(2.0 - frand()); } break; + } +} + +int main(void){ + /* sizes: small for exhaustive tie detection up to near-production scale */ + int sizes[] = {1, 2, 8, 64, 257, 1519}; /* 1519 ~= V/100 of GLM-5.2 */ + double nucs[] = {0.001, 0.5, 0.9, 0.999}; /* tight -> almost-everything */ + int n_shapes = 5; + + /* temperature used by dist_build: pick a normal serving value */ + g_temp = 0.7f; + + int cases = 0; + for (size_t si = 0; si < sizeof(sizes)/sizeof(sizes[0]); si++){ + int V = sizes[si]; + /* dist_build allocates g_pbuf/g_pidx ONCE and reuses them (single-V invariant in + * real serving, where V is the constant model vocab). This sweep varies V, so free + * and force a reallocation per size -- otherwise a later, larger V would overflow + * the buffer sized for the first (smallest) V. */ + free(g_pbuf); g_pbuf = NULL; free(g_pidx); g_pidx = NULL; + float *lo = malloc((size_t)V * sizeof(float)); + for (int shape = 0; shape < n_shapes; shape++){ + fill_shape(lo, V, shape); + for (size_t ni = 0; ni < sizeof(nucs)/sizeof(nucs[0]); ni++){ + char label[32]; snprintf(label, sizeof(label), "size[%zu]/shape[%d]", si, shape); + const char *sn = (const char*[]){"uniform","peaked","geometric","plateau","sharptail"}[shape]; + check_case(label, V, nucs[ni], sn, lo); + cases++; + } + } + free(lo); + } + + /* guard-off path: g_nuc >= 1 must skip truncation entirely (full softmax kept) */ + { + int V = 256; float lo[256]; + for (int i = 0; i < V; i++) lo[i] = (float)(frand()*4 - 2); + g_nuc = 1.0f; dist_build(lo, V); + int nz = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) nz++; + if (nz != V){ fprintf(stderr, " FAIL [guard-off nuc=1.0]: %d/%d entries kept, expected all\n", nz, V); g_nfails++; } + else printf(" ok [guard-off nuc=1.0 keep=%d]\n", nz); + cases++; + + g_nuc = 0.0f; dist_build(lo, V); + nz = 0; for (int i = 0; i < V; i++) if (g_pbuf[i] != 0.f) nz++; + if (nz != V){ fprintf(stderr, " FAIL [guard-off nuc=0.0]: %d/%d entries kept, expected all\n", nz, V); g_nfails++; } + else printf(" ok [guard-off nuc=0.0 keep=%d]\n", nz); + cases++; + } + + /* extreme tie edge case: V=1, single token -> keep=1 regardless of nuc */ + { + float lo[1] = {5.f}; + g_nuc = 0.5f; dist_build(lo, 1); + if (g_pbuf[0] == 0.f || !(fabs((double)g_pbuf[0] - 1.0) < 1e-6)){ + fprintf(stderr, " FAIL [V=1]: g_pbuf[0]=%.9g, expected 1.0\n", (double)g_pbuf[0]); g_nfails++; + } else printf(" ok [V=1 keep=1]\n"); + cases++; + } + + printf("\ntest_topp: %d cases run, %d failure(s)\n", cases, g_nfails); + if (g_nfails){ printf("test_topp: FAIL\n"); return 1; } + printf("test_topp: ok\n"); + return 0; +} diff --git a/c/tools/convert_olmoe.py b/c/tools/convert_olmoe.py index dd45806..dc25f6a 100644 --- a/c/tools/convert_olmoe.py +++ b/c/tools/convert_olmoe.py @@ -3,7 +3,10 @@ Downloads or converts a local OLMoE checkpoint (e.g., allenai/OLMoE-1B-7B-0125-Instruct). Dense weights stay as-is (engine reads BF16/F16 → F32 on load). -Expert weights get row-wise int8 quantization with float32 scales. +Expert weights get row-wise symmetric quantization to --ebits bits (default 4) +with float32 scales. Storage stays one value per int8 byte regardless of bits, +matching the engine's expert layout (olmoe.c quantize_rows) — for 4 bits the +values are simply confined to [-8, 7] with scales computed against qmax=7. Usage: python tools/convert_olmoe.py --repo allenai/OLMoE-1B-7B-0125-Instruct --out ./olmoe_i4 @@ -29,12 +32,21 @@ except ImportError as exc: EXPERT_KEY_RE = r"model\.layers\.\d+\.mlp\.experts\.\d+\.(gate_proj|up_proj|down_proj)\.weight" -def quantize_row(w: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: - """Row-wise int8 quantization. Returns (int8_weights, float32_scales).""" +def quantize_row(w: torch.Tensor, bits: int = 8) -> tuple[torch.Tensor, torch.Tensor]: + """Row-wise symmetric quantization to `bits` (2..8). + + Returns (int8_weights, float32_scales). Storage is one value per int8 byte + for every bit width — the engine dequantizes as q*scale and never assumes + the full int8 range — mirroring olmoe.c quantize_rows(): + qmax = 2**(bits-1) - 1 (8 -> 127, 4 -> 7, 2 -> 1) + scale = amax(|w|, row) / qmax + q = clamp(round(w / scale), -qmax-1, qmax) + """ + qmax = (1 << (bits - 1)) - 1 w_f32 = w.float() row_max = w_f32.abs().amax(dim=1, keepdim=True).clamp(min=1e-12) - scales = row_max / 127.0 - q = (w_f32 / scales).round().clamp(-128, 127).to(torch.int8) + scales = row_max / qmax + q = (w_f32 / scales).round().clamp(-qmax - 1, qmax).to(torch.int8) return q, scales.squeeze(1) @@ -50,9 +62,12 @@ def main(): src.add_argument("--model", help="Local HF checkpoint directory") ap.add_argument("--out", required=True, help="Output directory for int4 model") ap.add_argument("--ebits", type=int, default=4, - help="Expert quant bits (4 or 8, default 4)") + help="Expert quant bits (2..8, default 4)") args = ap.parse_args() + if not 2 <= args.ebits <= 8: # storage is int8_t; engine rejects the same range (olmoe.c) + sys.exit(f"--ebits must be 2..8 (got {args.ebits})") + if args.repo: from huggingface_hub import snapshot_download from huggingface_hub.errors import LocalEntryNotFoundError @@ -96,7 +111,7 @@ def main(): for name, tensor in tensors.items(): if is_expert_weight(name): expert_count += 1 - q, scales = quantize_row(tensor) + q, scales = quantize_row(tensor, args.ebits) total_expert_f32 += tensor.numel() * tensor.element_size() total_expert_q += q.numel() * 1 + scales.numel() * 4 out_tensors[name] = q @@ -109,7 +124,7 @@ def main(): ratio = total_expert_q / max(total_expert_f32, 1) * 100 print(f"ok") - print(f"\nDone. {expert_count} expert tensors quantized.") + print(f"\nDone. {expert_count} expert tensors quantized to int{args.ebits}.") print(f"Expert storage: {total_expert_f32/1e9:.1f} GB -> {total_expert_q/1e9:.1f} GB ({ratio:.0f}%)") print(f"Model ready at: {out}") print(f"\nRun: SNAP={out} ./olmoe.exe 32 4 16") diff --git a/c/tools/quant_ablation.py b/c/tools/quant_ablation.py index 689643a..dd08715 100644 --- a/c/tools/quant_ablation.py +++ b/c/tools/quant_ablation.py @@ -118,7 +118,7 @@ def quantize_param(w, bits, group, rot=False, e8=""): def _grid_or_e8(x, bits, group, e8): if e8: - return _quant_e8(x.float(), group, ball=(e8 == "-e8")) + return _quant_e8(x.float(), group, bits, ball=(e8 == "-e8")) return _quant_last_dim(x, bits, group) @@ -169,8 +169,15 @@ def _e8_ball(y, r2=10.0): return p -def _quant_e8(x, group, ball): - """Blocks of 8 along the input dim; per-group scale by MSE search over RMS multiples.""" +_E8_R2_REPORTED = set() +def _e8_radius(bits): + # E8 lattice: points within |p|^2<=r2 grow ~r2^4, so +1 bit (x256 codebook) needs r2 x4. + # Anchor: r2=10 is the ~2^16 E8P ball (2 bits over 8 dims). Scale from there. + return 10.0 * (4.0 ** (bits - 2)) + +def _quant_e8(x, group, bits, ball): + """Blocks of 8 along the input dim; per-group scale by MSE search over RMS multiples. + ball=True clamps to the rate-scaled E8 ball for `bits`; ball=False is the unbounded ideal.""" if x.shape[-1] % 8: raise SystemExit(f"-e8 needs input dim divisible by 8 (got {x.shape[-1]})") g = group or x.shape[-1] @@ -183,7 +190,11 @@ def _quant_e8(x, group, ball): for k in (0.5, 0.7, 0.9, 1.1, 1.4, 1.8, 2.4): s = rms * k yb = (xg / s).reshape(-1, g // 8, 8) - p = _e8_ball(yb) if ball else _e8_nearest(yb) + p = _e8_ball(yb, _e8_radius(bits)) if ball else _e8_nearest(yb) + if ball and bits not in _E8_R2_REPORTED: + _E8_R2_REPORTED.add(bits) + import sys as _sys + _sys.stderr.write(f"[e8] bits={bits}: ball r2={_e8_radius(bits):.1f}\n") out = (p.reshape(-1, g) * s) err = (out - xg).pow(2).sum(-1, keepdim=True) if best_err is None: diff --git a/docs/ENVIRONMENT.md b/docs/ENVIRONMENT.md index 4bf2d55..535288a 100644 --- a/docs/ENVIRONMENT.md +++ b/docs/ENVIRONMENT.md @@ -2,7 +2,7 @@ Reference for the environment variables read by the colibrì engine. -**Generated from `upstream/dev @ 6d3ed7e`** by scanning every `getenv()` site in `c/glm.c`. Defaults and behavior are taken from the source; see [MAINTAINING-DOCS.md](MAINTAINING-DOCS.md) to regenerate this after the code changes. +**Generated from `dev @ d5327e2`** by scanning every `getenv()` site in `c/glm.c` and the other C sources (`c/olmoe.c`, `c/backend_cuda.cu`, `c/backend_metal.mm`). Defaults and behavior are taken from the source; see [MAINTAINING-DOCS.md](MAINTAINING-DOCS.md) to regenerate this after the code changes. ## Which program reads these? @@ -43,6 +43,7 @@ Format: `VAR` — default — effect. | `URING` | `0` (off) | Linux-only queued expert I/O. `URING=1` implies `PIPE=1`, forces cold reads through io-wq (`IOSQE_ASYNC`), replaces blocking loader pthreads and spin waits with batched SQEs/CQEs, and batches `PILOT_REAL` loads on a separate ring. Use `DIRECT=1` for cold NVMe to avoid page-cache copy/readahead limits. Fails clearly if the kernel denies io_uring; incompatible with `COLI_MMAP=1`. | | `DIRECT` | `0` (off) | Use `O_DIRECT`/unbuffered reads for expert slabs. Helps sustained NVMe; keeps the zero-copy GPU path. | | `COLI_NO_OMP_TUNE` | off | **Kill-switch** for the OpenMP hot-thread tuning (`OMP_WAIT_POLICY=active` spin + proc-bind). Set `=1` when the CPU is mostly waiting on the GPU (Metal) so spin doesn't steal the shared power budget. | +| `COLI_NUMA` | off (Linux only) | `COLI_NUMA=1` interleaves expert slabs across NUMA nodes via `mbind` (raw syscall, no libnuma). Helps multi-socket hosts (+7–40% expert matmul); silent no-op on single-node or non-Linux. | | `MLOCK` | `-1` (auto: on for macOS) | Wire the streamed expert cache into physical RAM (`mlock`) to dodge the memory compressor. `0` off, `1` force. | | `CAP_RAISE` | `1` (on) | Let the engine raise the expert-cache cap above `topk` when RAM allows (bigger batches). `0` fixes the cap. | | `PREFETCH` | `0` | Prefetch depth for streamed experts. | @@ -54,16 +55,26 @@ Format: `VAR` — default — effect. | `PILOT` | `0` (off) | Router-piloted cross-layer expert prefetch. | | `PILOT_REAL` | `0` (off) | Value-preserving real cross-layer prefetch loads (`PILOT_REAL=1` opts in). | | `PILOT_K` | `6` if `PILOT_REAL` else `8` | Number of experts the pilot prefetches per step. | +| `PILOT_TWO` | `0` (off) | Two-step shared-expert-corrected router prediction for the pilot. | +| `COUPLE` | unset | Path to a coupling-score file driving cross-layer expert prefetch (#176). When set, `couple_load` reads it. | +| `COUPLE_K` | `8` | Top-K coupled experts per layer when `COUPLE` is set. | +| `COUPLE_D` | `1` | Coupling lookahead depth (`1` or `2`) when `COUPLE` is set. | | `CACHE_ROUTE` | `0` (off) | Opt-in max-rank cache-aware MoE routing (pin∪LRU prefer within top-M). See [CACHE_ROUTE.md](CACHE_ROUTE.md). | | `ROUTE_J` | `2` | Sacred top ranks always taken when `CACHE_ROUTE=1`. | | `ROUTE_M` | `12` | Max-rank window for resident preference when `CACHE_ROUTE=1`. | | `ROUTE_P` | `0` | Cumulative mass window for CACHE_ROUTE (`0` = fixed M). | | `ROUTE_ALPHA` | `1` | Scale gate mass of substituted experts before renorm (`1` = off). | | `ROUTE_AGREE` | auto | Overlap% + KL vs true top-K; auto-on when `CACHE_ROUTE=1`. | +| `ROUTE_TRACE` | unset | If set to a path, logs every routing decision there (testing/analysis). | | `ABSORB` | `-1` (auto: absorbed for S≤4) | MLA attention absorption mode. | | `IDOT` | `1` | Integer dot-product kernel. `IDOT=0` uses exact f32 kernels (for A/B numerical checks). | | `COLI_POLICY` | `quality` | Resource policy: `quality`, `balanced`, or `experimental-fast`. | | `PROF` | `0` (off) | Performance profile: a startup header (machine + effective config), then per run — or per turn in serve mode, on stderr — forward-latency percentiles (p50/p90/p99/max), expert-I/O totals and cache-tier fill, phase shares of wall time, and a verdict naming the knob most likely to help on this machine. Output is additive; `PROF` unset changes nothing. | +| `COLI_NO_FUSED_PAIR` | `0` (off) | `=1` disables the fused-pair matmul kernel. | +| `DISK_SPLIT` | `0` (off) | `=1` splits the reported disk-load time across the draft/absorb/forward phases in stats. | +| `I4S` | unset | Engage the int4 `IDOT` kernel only for batch `S>=` (testing). | +| `SPEC_PIN` | `1` (on) | Speculation gate mode. `0` reverts to the legacy S-dependent speculation gates (#163). | +| `COLI_RAM_OVERCOMMIT` | off | `=1` overrides the "projected peak > MemAvailable → exit(2)" guard so a run that risks kernel OOM-kill is allowed to proceed. | --- @@ -77,7 +88,22 @@ Format: `VAR` — default — effect. | `CUDA_EXPERT_GB` | `0` | VRAM budget (GB) for caching experts on the GPU. | | `CUDA_RELEASE_HOST` | auto (`1` if >1 device) | Release host-side copies after upload. | | `COLI_CUDA_ATTN` | off | Run S≤4 attention on the GPU. | +| `COLI_CUDA_ATTN_SHARD` | off | `=1` splits KV-b heads across devices during attention load (multi-GPU). | | `COLI_CUDA_PROFILE` | off | Emit CUDA timing. | +| `COLI_CUDA_PIPE` | `0` (off) | `1` engages the multi-step attention pipeline; `2` enables the pipe2 path. | +| `COLI_CUDA_PIPE_SHARD` | off | `=1` runs the multi-device P2P head-shard attention path (opt-in for NVLink topologies; serializes ~95 MB/layer over a star PCIe topology). | +| `COLI_CUDA_PIPE_S_MIN` | `1` single-GPU, `8` multi-GPU | Minimum prefill batch S to engage the pipe2 CUDA path. | +| `COLI_CUDA_MTP` | `0` (off) | `=1` opts into MTP speculation under CUDA (off by default: cold streaming experts run on CPU where the fused-pair/IDOT kernels diverge in FP order, collapsing draft acceptance, #163/#292). | +| `COLI_CUDA_ASYNC` | on | `=0` forces synchronous `cudaMemcpy` instead of async + pinned host staging. | +| `COLI_CUDA_DUAL_PROJ` | on | `=0` issues gate+up as two separate launches instead of one fused `grouped_hidden_w4_dual`. | +| `COLI_CUDA_W4_PACKED` | on | `=0` disables the grouped packed-int4 path. | +| `COLI_CUDA_TC_INT4` | off | `=1` uses the W4A4 WMMA Tensor Core path (when all expert tensors are int4 and dims divide). | +| `COLI_CUDA_TC_MIN_ROWS` | `8` | Min rows-per-expert to engage the W4A4 Tensor Core path. | +| `COLI_CUDA_TC_W4A16` | off | `=1` uses the lossless W4A16 Tensor Core path (compute capability ≥7). | +| `COLI_CUDA_TC_W4A16_MIN` | `16` | Per-expert row threshold above which W4A16 TC tiles dispatch (smaller batches fall back to the naive kernel). | +| `COLI_CUDA_SHARED_W4A16` | off | `=1` uploads shared-expert weights and runs the shared-MLP W4A16 Tensor Core kernel. | +| `COLI_CUDA_SHARED_W4A16_MIN_ROWS` | `32` | Min row count to engage the shared-MLP W4A16 kernel. | +| `COLI_METAL_UNTRACKED` | off (Metal only) | `=1` sets `MTLResourceHazardTrackingModeUntracked` on Metal buffers (reduces hazard-tracking overhead). | --- @@ -89,8 +115,11 @@ These are for testing, benchmarking, or internal use — not part of the everyda |---|---|---| | `SPEC` | `1` | Speculative decoding on/off. | | `DRAFT` | `-1` (auto: 3 with MTP, else 0) | Number of speculative draft tokens per step. | -| `GRAMMAR` | unset | Path to a GBNF grammar file to constrain generation. | +| `GRAMMAR` | unset | Path to a GBNF grammar file to constrain generation. Takes precedence over `SCHEMA`. | +| `SCHEMA` | unset | Path to a JSON-Schema file compiled to GBNF to constrain generation (consulted only when `GRAMMAR` is empty). | | `GRAMMAR_DRAFT` | unset | Max grammar-forced draft span length. | +| `EXPERT_BUDGET` | `0` (off) | Cap experts loaded per layer (MoE-Spec). **Quarantined:** silently forced to `0` unless `EXPERT_BUDGET_EXPERIMENTAL` is set — every tested value is either no faster or incoherent (issue #303). | +| `EXPERT_BUDGET_EXPERIMENTAL` | unset | Setting it (any value) allows `EXPERT_BUDGET>0` to actually take effect (expect garbage, #294). | | `DSA` | on | Dynamic Sparse Attention indexer. `DSA=0` disables. | | `DSA_FORCE` | `0` | Force the DSA path on. | | `DSA_TOPK` | model value | Override the DSA index top-k (testing). | @@ -101,11 +130,15 @@ These are for testing, benchmarking, or internal use — not part of the everyda | `PIN_FILL` | `0` | Fill the pinned store even without usage data. | | `MTP_DEBUG` / `MTP_PRENORM` / `MTP_SWAP` | off | MTP head debugging / ablations. | | `STATS` | unset | Write an expert-usage histogram to `STATS=` at end of run. | +| `TOKENS` | unset | If set, dumps generated token ids to stderr for A/B comparison. | | `SCORE` | unset | Scoring/eval mode over `SCORE=`. | +| `SCORE_PREFIX` | on | If unset or `≠0`, prepends `[gMASK]` to scoring contexts (GLM-family only). | +| `REPIN_VERBOSE` | off | If set, prints per-swap `[REPIN]` diagnostics during VRAM repin. | | `REF` / `REF_FORCE` | `ref_glm.json` | Reference-output comparison mode. | | `REPLAY` | unset | Replay mode. | | `TF` | unset | Teacher-forcing mode. | | `CHAT_TEMPLATE` | `1` | Apply the GLM chat template (`0` = raw prompt). | +| `PPL` | off (`olmoe.c` only) | `PPL=1` enters teacher-forced NLL/perplexity meter mode in the OLMoE sister engine. | --- @@ -136,7 +169,7 @@ These are read by the Python programs (not the `glm` engine), so they don't appe - `SNAP` — model snapshot directory (required by `glm`; set from `--model`). - `SERVE`, `SERVE_BATCH` — select serve / batched-serve mode. -- `PROMPT` — one-shot text mode. +- `PROMPT` — one-shot text mode (the engine also honors `COLI_PROMPT`, preferred cross-platform; `PROMPT` is ignored on Windows if it contains cmd.exe `$`-metacharacters). - `COLI_OMP_TUNED` — internal sentinel guarding the OMP re-exec (see `COLI_NO_OMP_TUNE`); not user-facing. --- diff --git a/docs/WINDOWS.md b/docs/WINDOWS.md new file mode 100644 index 0000000..52d83cd --- /dev/null +++ b/docs/WINDOWS.md @@ -0,0 +1,100 @@ +# Windows 11 native install — a complete walkthrough (no WSL) + +A start-to-finish, reproducible path from a fresh Windows 11 machine to GLM-5.2 generating tokens, with the GPU tier. Every step and every failure mode below was hit and verified on real hardware: Core Ultra 9 285K (AVX-VNNI) / RTX 5080 (sm_120) / 128 GB RAM / Windows 11 24H2 (issue #306). Steps are ordered so the long downloads run while you build. + +## 0. What you need + +| Piece | Why | Get it | +|---|---|---| +| git, Python 3 | clone + `coli` launcher | winget / python.org | +| MinGW-w64 gcc + make | builds the engine (MSVC can't) | `scoop install mingw-winlibs`, MSYS2, or portable **w64devkit** (no admin, unzip and go) | +| CUDA Toolkit ≥ 12.8 | GPU tier; ≥12.8 required for Blackwell/sm_120 | `winget install Nvidia.CUDA` | +| MSVC Build Tools (C++ workload) | nvcc's host compiler for the CUDA DLL | `winget install Microsoft.VisualStudio.2022.BuildTools` + "Desktop development with C++" | +| ~400 GB free on a local NVMe | the int4 model (~370–384 GB) | NTFS is fine; **never** a network mount | + +RAM: 16 GB minimum, more = bigger expert cache = faster. The build itself needs none of the CUDA/MSVC pieces — do the CPU build first, add the GPU tier later. + +## 1. Start the model download first (it's the long pole) + +```powershell +python -m pip install -U "huggingface_hub[hf_transfer]" +$env:HF_HUB_ENABLE_HF_TRANSFER = "1" +hf download --local-dir D:\glm52_i4 +``` + +Use the container recommended in the README (with **int8 MTP heads** — int4 heads silently give 0% draft acceptance). The download is resumable: if it stops, rerun the same command. Expect hours; everything below fits inside them. + +## 2. Build the engine (CPU) + +From a normal PowerShell, in the repo's `c\` directory: + +```powershell +make glm.exe ARCH=native # ARCH=native unlocks AVX-VNNI on Alder Lake+/Arrow Lake +make iobench.exe # disk benchmark, useful before committing to the download +``` + +Warnings about `#pragma comment` and unused variables are normal (MSVC-isms gcc ignores). The engine banner should print `idot: avx-vnni` on VNNI-capable CPUs — if it says avx2, you built without `ARCH=native`. + +### ⚠️ Smart App Control will block your fresh binary + +On Windows 11 machines with **Smart App Control** enforced (`VerifiedAndReputablePolicyState = 1`), running your self-compiled `glm.exe` fails with: + +``` +Program 'glm.exe' failed to run: An Application Control policy has blocked this file +``` + +This is not Defender and not Mark-of-the-Web — SAC blocks *all* unsigned, unknown binaries, which includes anything you compile yourself. **Fix:** Windows Security → App & browser control → Smart App Control settings → **Off**, then **reboot** (the policy only reloads on restart). Note SAC is one-way: re-enabling later requires resetting Windows. If the settings page is missing, the registry equivalent is setting `HKLM:\SYSTEM\CurrentControlSet\Control\CI\Policy\VerifiedAndReputablePolicyState` to `0` (admin PowerShell), then rebooting. Check your current state before touching anything: + +```powershell +(Get-ItemProperty "HKLM:\SYSTEM\CurrentControlSet\Control\CI\Policy").VerifiedAndReputablePolicyState +# 0 = off, 1 = enforced, 2 = evaluation +``` + +## 3. Build the CUDA DLL (GPU tier) + +nvcc needs MSVC as host compiler, so this one step must run from a shell with the MSVC environment: open **"x64 Native Tools Command Prompt for VS 2022"** from the Start menu (plain PowerShell will fail the `cl` check). Then: + +```cmd +make cuda-dll CUDA_ARCH=sm_120 # match your GPU: sm_120 Blackwell, sm_89 Ada, ... +make glm.exe CUDA_DLL=1 ARCH=native # relink host with the runtime loader +``` + +Two pitfalls, both fixed on current `dev` (#314) but worth knowing on older checkouts: + +- **Spaces in `CUDA_HOME`** (`C:\Program Files\...`) used to break the recipe → fixed; nvcc now comes from PATH and `"$(NVCC)"` is quoted. +- **`make glm.exe CUDA_DLL=1` after a CPU-only build** used to report `up to date` and silently keep the CPU-only binary (GPU tier never engages, no error). Current `dev` has a build-config stamp that forces the relink. On older trees: delete `glm.exe` first. + +Sanity check: first GPU run should print `[CUDA] device 0: , ... sm_XX` and `[CUDA] mode: routed experts + resident dense tensors`. + +## 4. First run + +```powershell +cd \c +$env:OMP_NUM_THREADS = "" +python coli run "Explain what a mixture-of-experts model is." --model D:\glm52_i4 --ngen 48 +``` + +The first run is cold — expect the profile to be dominated by `expert-disk` while the cache warms; hit rate climbs run over run. GPU tier on top: + +```powershell +$env:COLI_CUDA="1"; $env:COLI_GPU="0"; $env:CUDA_DENSE="1"; $env:CUDA_EXPERT_GB="4" +python coli run "..." --model D:\glm52_i4 --ngen 64 +``` + +Size `CUDA_EXPERT_GB` so dense (~10 GB) + experts + working set stays under your VRAM. Note MTP speculation is off by default under CUDA (#293, float-accumulation divergence between draft and verify) — `COLI_CUDA_MTP=1` opts back in. + +## 5. Reference numbers from this walkthrough's hardware + +285K / RTX 5080 / 128 GB / NVMe at 5.85 GB/s random-read (19 MB blocks, `iobench`): 0.26 tok/s cold CPU → 0.30 warm CPU (MTP 2.2–2.3 tok/forward) → 0.42 tok/s GPU tier + auto-pin, expert hit 66%, ~65% of wall time in expert-disk. Disk-bound is the expected shape at ~25% expert residency — a faster disk and more RAM move the floor, the GPU moves the compute. + +## Quick failure index + +| Symptom | Cause | Fix | +|---|---|---| +| `An Application Control policy has blocked this file` | Smart App Control | §2 — turn SAC off + **reboot** | +| `cuda-dll ... Error 1` immediately | old tree: spaced CUDA_HOME / MSVC rejects `-Wextra` | update to current `dev` (#314) | +| `glm.exe is up to date` but GPU never engages | old tree: stale CPU-only binary | update to `dev`, or delete `glm.exe` and rebuild | +| `cl.exe (MSVC) not in PATH` | built from plain PowerShell | use the x64 Native Tools prompt | +| `nvcc fatal: unsupported gpu architecture 'sm_120'` | CUDA < 12.8 | install CUDA 12.8+ | +| MTP `0% (0/0)` on CPU path | int4 MTP heads in the container | use the int8-MTP container | +| MTP `draft=0` under CUDA | intended default since #293 | `COLI_CUDA_MTP=1` to opt in | diff --git a/docs/api.md b/docs/api.md new file mode 100644 index 0000000..d3b09f1 --- /dev/null +++ b/docs/api.md @@ -0,0 +1,110 @@ +# OpenAI-compatible API, KV contexts & web UI + +## `coli serve` + +`coli serve` keeps one model process loaded and exposes a text-only +OpenAI-compatible HTTP API. The gateway uses only the Python standard library; +inference still runs in the same dependency-free C engine. + +```bash +cd c +COLI_MODEL=/nvme/glm52_i4 COLI_API_KEY=local-secret ./coli serve \ + --host 127.0.0.1 --port 8000 --model-id glm-5.2-colibri + +curl http://127.0.0.1:8000/v1/chat/completions \ + -H 'Authorization: Bearer local-secret' \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "glm-5.2-colibri", + "messages": [{"role": "user", "content": "Hello"}], + "stream": true + }' +``` + +Implemented endpoints are `GET /v1/models`, `GET /v1/models/{model}`, +`POST /v1/chat/completions`, and legacy `POST /v1/completions`. Chat and +completion requests support JSON responses, SSE streaming, usage counts, +`max_tokens`/`max_completion_tokens`, `temperature`, and `top_p`. The extension +`enable_thinking: true` enables GLM-5.2's reasoning block; the standard +`reasoning_effort` field also enables it unless set to `none`. + +The server is deliberately text-only and serves one generation at a time: the +744B model stays in one persistent process, so concurrent HTTP requests queue +instead of loading duplicate model copies. Tools, image/audio input, custom +stop sequences, log probabilities, and token penalties return an explicit error +rather than being silently ignored. The default bind address is localhost; set +`COLI_API_KEY` before exposing the server beyond the machine. + +Browser access from the Vite development server and Tauri local origins is +enabled by default. Repeat `--cors-origin https://your-ui.example` to allow +another exact origin, or use `--cors-origin '*'` only on a trusted local +network. + +The engine owns its KV contexts, so HTTP generation uses a bounded FIFO +admission queue instead of pretending to run unsafe parallel sequences. +Configure it with `--max-queue N` (default 8) and `--queue-timeout SECONDS` +(default 300), or the `COLI_MAX_QUEUE` / `COLI_QUEUE_TIMEOUT` environment +variables. Saturated and timed-out requests receive OpenAI-shaped HTTP 429 +errors before streaming headers are sent. `GET /health` exposes +active/queued/completed/rejected counters, and successful generation responses +include `x-colibri-queue-wait-ms`. + +## Isolated KV contexts + +`coli serve --kv-slots N` allocates up to 16 independent sequence contexts. +Requests select one with the optional integer `cache_slot` field; ordinary +OpenAI clients omit it and keep the original slot 0 behavior. + +```json +{ + "model": "glm-5.2-colibri", + "messages": [{"role": "user", "content": "Continue this conversation"}], + "cache_slot": 1 +} +``` + +Each slot owns its token history, compressed MLA/DSA KV memory, MTP window, and +crash-safe persistence file (`.coli_kv`, `.coli_kv.1`, ...). The engine matches +each request's tokenized prompt against the slot's history and reuses the common +KV prefix, so stateless HTTP turns keep their cache across requests and even +across engine restarts. Use `COLI_KV_SLOTS=N` as the environment equivalent. +Start small: at the default 4096-token context, every slot costs hundreds of MB. + +## Web dashboard + +One command serves the OpenAI-compatible API **and** the web console on the +same port, then opens your browser when the engine is ready: + +```bash +cd web && npm install && npm run build # once +./coli web --model +``` + +What you get: + +- **Chat** with live metrics: a flashing token counter while generating, then + tok/s, time-to-first-token, prompt→completion counts and queue wait; +- **Runtime panel**: your hardware (CPU, GPUs + VRAM, RAM, cores), the + scheduler, and the live expert-tier bar — how many of the 19,456 experts sit + in VRAM / RAM / disk right now; +- **Brain**: the whole model as a 76×256 cortex, one cell per expert. Colour = + tier, brightness = routing heat, and the experts routed in each turn flash + white and decay — you watch the model think. Hover any cell for its tier, + heat and [measured topic affinity](https://github.com/JustVugg/colibri/issues/175); +- **Atlas**: the measured expert atlas as a 3-D galaxy (publish `experts.json` + from `tools/expert_atlas/analyze.py --web`). + +The dashboard talks to the engine over a small line protocol and plain JSON +endpoints — nothing heavier than the engine itself. `web/` is a pure OpenAI-API +client (React + TypeScript) and also works against any other compatible +endpoint; the terminal `coli chat` remains the first-class interface. + +The layout is responsive down to phone widths, and the sidebar carries the full +telemetry stack — hardware, scheduler, tier bar, per-turn time breakdown, tok/s +trend and per-GPU expert counts: + +

+ the dashboard on a phone-sized viewport +    + the telemetry sidebar +

diff --git a/docs/benchmarks.md b/docs/benchmarks.md new file mode 100644 index 0000000..f9c83f2 --- /dev/null +++ b/docs/benchmarks.md @@ -0,0 +1,140 @@ +# Benchmarks & measured numbers + +Everything on this page is a measurement, not a promise. If you run colibrì on +hardware not listed here, **please open an issue with your numbers** — real +datapoints are what move this project. + +## Reference numbers (the original dev box: WSL2, 12 cores, 25 GB RAM, NVMe via VHDX) + +Detailed GPU experiment: [GLM-5.2 on 6× RTX 5090](experiments/glm52-6x5090-2026-07-12.md) — +full expert residency across VRAM+RAM reaches **6.84 tok/s** single-request decode. + +| metric | value | +|---|---| +| model on disk (int4 container) | ~370 GB | +| resident RAM (dense, int4) | 9.9 GB | +| load time | ~30 s | +| peak RSS during chat | ~20 GB (auto-capped) | +| cold decode cost | ~11 GB disk reads/token (75 layers × 8 experts) | +| disk ceiling (this dev box's drive) | ~1 GB/s → ~0.05–0.1 tok/s cold | +| MTP speculation (int8 head) | 2.2–2.8 tok/forward measured ([#8](https://github.com/JustVugg/colibri/issues/8)) | + +This is not fast. It is a 744B frontier-class model **answering correctly on a +machine that costs less than one H100 fan**. Warm cache, pinned hot experts and +MTP push the useful-response latency down considerably; the physics of the disk +does the rest. + +### SSD note + +Cold starts are heavy on random reads (~11 GB/token), but reads don't +meaningfully wear an SSD — colibrì's streaming is read-only. The real concerns +under heavy use are (1) **swap traffic** if the system runs out of RAM (writes +do wear the drive — keep a sane `--ram` budget; colibrì's auto-budget is designed +to stay clear of swap) and (2) **sustained thermals**: hours at full read duty +cycle will heat cheaper drives. Monitor drive temperature and health. + +## Test your machine, in order + +```bash +cd c && ./setup.sh # build + architecture self-test (expects 32/32) + +# 1) measure YOUR disk the way the engine uses it (parallel 19 MB random reads): +gcc -O2 -fopenmp iobench.c -o iobench +./iobench /path/to/glm52_i4/out-00069.safetensors 19 64 8 0 # buffered, 8 threads +./iobench /path/to/glm52_i4/out-00069.safetensors 19 64 8 1 # O_DIRECT (bypass cache) +# Caveat (#86): iobench reads a bounded ~1 GB shard, so buffered reads on a big-RAM box +# report the PAGE CACHE, not the disk. Use the O_DIRECT run (arg 1) for a true number, and +# run it on a shard you haven't touched this session (a prior buffered run caches its pages). +# On macOS there is no O_DIRECT — iobench uses F_NOCACHE, which stops *new* caching but can't +# evict pages a prior buffered run already resident-mapped, so a macOS "O_DIRECT" figure right +# after a buffered run still reads cache. Reboot or use a fresh shard for a real cold read. + +# 2) chat; watch the per-turn stats line (tok/s, expert hit-rate, RSS): +COLI_MODEL=/path/to/glm52_i4 ./coli chat + +# 3) record expert usage, then pin the hottest experts in your spare RAM: +STATS=stats.txt ./coli chat +PIN=stats.txt PIN_GB=20 ./coli chat # scale PIN_GB to your free RAM + +# 4) quality benchmarks (MMLU/HellaSwag/ARC): +./coli bench +``` + +## Back-of-envelope predictions + +Decode is disk-bound: a cold token costs ~11.4 GB of expert reads; MTP +speculation roughly halves the effective cost *once the cache is warm*; RAM +turns cold reads into free cache hits. + +| machine | expected | +|---|---| +| the dev box (WSL2 VHDX, ~1 GB/s, 25 GB RAM) | ~0.05–0.1 tok/s cold — proven baseline | +| native Linux, PCIe4 NVMe (~3–5 GB/s random), 32 GB | ~0.5–1 tok/s | +| PCIe5 NVMe or 2×NVMe RAID0 (~8–12 GB/s), 64 GB (PIN ~40 GB of hot experts) | ~2–4 tok/s | +| 128–256 GB RAM, 12 cores (hot experts cached) | ~2–4 tok/s — matmul-bound: ~80 GFLOP/token vs ~250 GFLOP/s of our AVX2 kernels | +| same RAM + 24–32 cores, or AVX-512/VNNI kernels | ~5–15 tok/s — interactive; kernel work is the multiplier | + +These are estimates, not measurements. + +## Community benchmarks (measured) + +Real numbers from real machines, stock build (`setup.sh`, gcc 13), greedy decoding, `--ngen 32`, MTP active: + +| machine | disk (iobench, 19 MB × 64, 8 threads) | config | measured | +|---|---|---|---| +| Intel Core Ultra 7 270K Plus (24 threads) · WSL2 · 24 GB RAM · NVMe VHDX ([#2](https://github.com/JustVugg/colibri/issues/2)) | 1.96 GB/s buffered · 2.74 GB/s O_DIRECT | default | 0.07 tok/s · expert hit 3–4% · RSS 14.1 GB | +| 〃 | 〃 | `--topp 0.7` | **0.11 tok/s** · expert hit 11% · RSS 14.7 GB | +| Apple M5 Max (18 cores) · macOS · 128 GB unified · internal SSD ([#4](https://github.com/JustVugg/colibri/issues/4), [#5](https://github.com/JustVugg/colibri/issues/5)) | ~4 GB/s cold (the 14.2 GB/s reading was cache-influenced — see note) | default, MTP off | **1.06 tok/s** · expert hit 23% · RSS 21.8 GB | +| Apple M5 Max · macOS · 128 GB unified · 2 TB SSD · **Metal backend** ([#72](https://github.com/JustVugg/colibri/pull/72), [#87](https://github.com/JustVugg/colibri/issues/87)) | (macOS O_DIRECT figure unreliable — see note) | Metal on · `--ram 96` · 39.7 GB warm pin · MTP off | **1.83 tok/s** · expert hit 66% · warmed 1.11 → 1.83 over the run | +| 〃 · 46.9 GB pin (2.94M-selection history) · `--ram 110`, 1024-token run ([#103](https://github.com/JustVugg/colibri/issues/103)) | 〃 | Metal on (experts + attention) · MTP off | **2.06 tok/s** · hit 72.5% · coherent output | +| Mac Mini M4 Pro · macOS · **48 GB** unified · **Metal backend** ([#107](https://github.com/JustVugg/colibri/issues/107)) | 6.59 GB/s F_NOCACHE (fresh shard) | Metal on · `--ram 38` | **0.30 tok/s** (vs 0.18 CPU-only) | +| Epyc 9654 ES · Linux · 4x16GB DDR5-4800-rdimm · Samsung PCIe Gen3 x4 NVME SSD | — | `MTP=1 DIRECT=1` | 0.31 tok/s · expert hit 35% · RSS 21.52 GB | +| Ryzen AI 9 HX 370 (Framework 13) · Arch Linux · 128 GB · WD SN850X, BTRFS zstd ([#12](https://github.com/JustVugg/colibri/issues/12)) | — | int8 MTP head · `--cap 32` · 46.7 GB auto-learned PIN | **0.37 tok/s** · expert hit 66% · MTP acceptance 52% (2.59 tok/fw) · RSS 105 GB | +| Ryzen 9 9950X (32 threads) · Linux · 123 GB · Crucial P3 QLC Gen3 ([#31](https://github.com/JustVugg/colibri/issues/31)) | 1.51 GB/s buffered | default, 2 runs from cold | 0.10 tok/s · hit 53% · profile 66% disk | +| 〃 same machine, model moved to a Samsung 9100 PRO PCIe 5.0 ([#31](https://github.com/JustVugg/colibri/issues/31)) | **8.81 GB/s** O_DIRECT | 〃 (usage history retained) | **0.28 tok/s** · hit 57% · profile flips: 32% disk / **57% matmul** | +| Ryzen AI Max+ 395 (Framework Desktop) · Ubuntu · 128 GB LPDDR5x · Intel Optane 905p PCIe 3.0 ([#39](https://github.com/JustVugg/colibri/issues/39)) | 3.27 GB/s buffered | int8 MTP head · fresh history (pure LRU, auto-raised cap 65) | 0.16 tok/s · hit 57% · profile 49% disk / 47% matmul | +| 〃 five runs later — learned pin 47.6 GB ([#39](https://github.com/JustVugg/colibri/issues/39)) | 〃 | `--temp 0.7 --topp 0.7` | **0.40 tok/s** · hit 71% | +| Ryzen 7 9800X3D (16T) · WSL2 · 70 GB RAM · Samsung 9100 PRO PCIe 5.0 · RTX 5090 ([#101](https://github.com/JustVugg/colibri/issues/101)) | **10.51 GB/s** O_DIRECT | MTP off · learned pin 24 GB · hit 54% · OMP hot-team on | **0.41 tok/s** · disk-bound (36.5 s disk vs 24.0 s matmul) · **CUDA expert tier ≈ 0%** (AVX-512 CPU matches the 5090) · `--topp 0.7` → **0.52 tok/s** | +| EPYC 7443 (24C/48T, Zen3 AVX2) · Linux · **430 GB RAM** · NVMe RAID-Z1 via TrueNAS VM ([#104](https://github.com/JustVugg/colibri/issues/104)) | ~1 GB/s (VM overhead) | 77.5 GB pin · cap auto-raised to 194/layer · MTP off | **1.00 tok/s** · **hit 98%** · disk eliminated → **RAM-bandwidth + matmul bound** | +| Intel i5-12600K (10C/16T, AVX2) · **native Windows 11, no WSL** · 32 GB · MinGW GCC 16.1 ([#113](https://github.com/JustVugg/colibri/issues/113)) | buffered (no O_DIRECT on MinGW) | int8 MTP head · cold, small-RAM (cap ~2/layer) | **0.08 tok/s** · hit 3.7% · **MTP 57% acceptance** — first native-Windows datapoint | +| Ryzen 9 9950X3D2 (16C/32T, avx512-vnni) · native Linux · 121 GB · Samsung 9100 PRO **PCIe Gen5** · RTX 5090 (28 GB expert tier, 1475 pinned) ([#120](https://github.com/JustVugg/colibri/issues/120)) | **11.48 GB/s** O_DIRECT | `MTP=0 DIRECT=1 PIPE_WORKERS=16 PREFETCH=1` | **1.23 tok/s** | +| Ryzen AI Max+ 395 (Strix Halo, 16C/32T Zen5, avx512-vnni) · Arch Linux · 128 GB unified LPDDR5x · SK hynix P41 PCIe 4.0 ([#124](https://github.com/JustVugg/colibri/issues/124)) | — | `DIRECT=1 PIPE=1 --topp 0.7` · auto-pin | 0.06 cold → **1.10 tok/s** sustained · later **1.83 tok/s** on current dev with `DIRECT=1 PIPE=1 PILOT_REAL=1 PILOT_TWO=1` ([#200](https://github.com/JustVugg/colibri/issues/200)) | +| Intel Core Ultra 9 185H (16C/22T, avx-vnni) · **native Windows 11, no WSL** · 32 GB · Crucial P3 QLC NTFS · RTX 5070 Ti ([#128](https://github.com/JustVugg/colibri/issues/128), [#273](https://github.com/JustVugg/colibri/issues/273)) | — | int8 MTP head · warm cache · GPU-resident pipeline at decode | 0.03 cold → 0.5 warm CPU → **1.07 tok/s** with the pipe2 decode gate (#274) | +| Dell Pro Max GB10 (DGX Spark: Grace, **aarch64 i8mm/sve2**) · Linux · 121 GB unified LPDDR5x · GB10 sm_121 ([#136](https://github.com/JustVugg/colibri/issues/136), [#161](https://github.com/JustVugg/colibri/issues/161)) | **5.58 GB/s** O_DIRECT | int8 MTP head · warm cache | 0.50 tok/s warm · **2.4 tok/s full-k8**, **3.33 tok/s** with `CACHE_ROUTE` (#199) | +| **6 × RTX 5090 · dual Xeon Silver 4510 · 251 GB** (author's rig, [experiment log](experiments/glm52-6x5090-2026-07-12.md)) | NVMe | `CUDA_EXPERT_GB=auto PIN_GB=all` full residency · `COLI_CUDA_PIPE=2 TC_W4A16` · DRAFT=0 | **5.8–6.8 tok/s** decode · TTFT ~13 s · hit 89–100% | + +### Takeaways + +With 24 GB of RAM the engine auto-caps the expert cache to 2 slots/layer, so +decode stays cold even on a fast disk — **on small-RAM machines the RAM cap, not +the disk, is the binding constraint**; `--topp 0.7` alone bought a clean 1.6× +end-to-end speedup. The 9950X pair is the cleanest bottleneck experiment: same +machine, same history, only the disk swapped — ×5.8 disk bandwidth bought ×2.9 +tokens, and the profile **flipped from 66% disk to 57% matmul**. But the +crossover depends on the CPU kernel: with OMP hot-team tuning on, an AVX-512 CPU +can match an RTX 5090 on expert matmul ([#101](https://github.com/JustVugg/colibri/issues/101)), +so **the GPU tier earns its VRAM only when the CPU is the weak link**. On +multi-socket hosts, NUMA placement is a further lever: interleaving the resident +weights across nodes measured **+13% (2-socket) and +40% (4-socket CPU-only)** +([#82](https://github.com/JustVugg/colibri/issues/82)) — but never blanket-interleave +a GPU host (measured 10× regression via the DMA staging pages). + +## Quality benchmark + +**Measured** ([#108](https://github.com/JustVugg/colibri/issues/108)): the int4 +container scored **62.5% mean acc_norm** on hellaswag/arc/mmlu (0-shot +log-likelihood, n=40) — but 0-shot MC scoring underserves a reasoning model, and +the OLMoE fp16-vs-int4 A/B under the same harness measured the pure quantization +cost at **-8.2pp**, concentrated on the hardest task (per-row int4 scales erode +the small logit margins hard questions depend on — grouped scales recover ~63% +of that loss, see [#225](https://github.com/JustVugg/colibri/issues/225)). The +scale-granularity/rotation/lattice ablation lives in +`tools/quant_ablation.py` ([#81](https://github.com/JustVugg/colibri/issues/81)). + +```bash +cd c +pip install tokenizers datasets +./coli bench # hellaswag, arc_challenge, mmlu — 40 questions each +./coli bench hellaswag --limit 200 # one task, more questions +./coli bench mmlu arc_challenge --ram 100 # pick tasks, set a RAM budget +``` diff --git a/docs/cuda.md b/docs/cuda.md new file mode 100644 index 0000000..fab477d --- /dev/null +++ b/docs/cuda.md @@ -0,0 +1,104 @@ +# CUDA backend (Linux) + +colibrì includes an opt-in CUDA backend for model-resident tensors. Streaming +experts deliberately remain on the original CPU path: copying an expert from +NVMe to the GPU on every use would only replace the disk bottleneck with a PCIe +bottleneck. Resident quantized tensors are uploaded lazily once and reused. + +```bash +cd c +make cuda-test CUDA=1 # q8/q4/q2/f32 kernel correctness +make CUDA=1 +# optional dense-path experiment (hot experts are configured below) +COLI_CUDA=1 COLI_GPU=0 CUDA_DENSE=1 SNAP=/nvme/glm52_i4 ./glm 64 4 4 +``` + +Requirements: Linux, an NVIDIA driver, and a CUDA Toolkit under +`/usr/local/cuda` (override with `CUDA_HOME=/path/to/cuda`). +`CUDA_ARCH=native` builds for the GPU in the current machine. Requesting CUDA +with a CPU-only binary, an invalid device, or an unavailable runtime fails at +startup instead of silently falling back. For Windows, see +[windows.md](windows.md) (runtime DLL path). + +## The VRAM expert tier + +A measured `PIN` profile promotes its hottest experts into a persistent VRAM +tier while keeping the rest in RAM: + +```bash +STATS=stats.txt SNAP=/nvme/glm52_i4 ./glm 64 4 4 # collect routing frequencies first +COLI_CUDA=1 COLI_GPU=0 CUDA_EXPERT_GB=16 \ +PIN=stats.txt PIN_GB=160 SNAP=/nvme/glm52_i4 ./glm 64 4 4 + +# multi-GPU expert tier, 150 GB total budget across six 32 GB devices +COLI_CUDA=1 COLI_GPUS=0,1,2,3,4,5 CUDA_EXPERT_GB=150 \ +CUDA_DENSE=1 PIN=stats.txt PIN_GB=300 RAM_GB=226 \ +SNAP=/nvme/glm52_i4 ./glm 64 4 4 + +# large-RAM host: fill safe VRAM, then keep every remaining expert in RAM +COLI_CUDA=1 COLI_GPUS=0,1,2,3,4,5 CUDA_EXPERT_GB=auto \ +CUDA_DENSE=1 COLI_CUDA_ATTN=1 PIN=stats.txt PIN_GB=all RAM_GB=auto \ +SNAP=/nvme/glm52_i4 ./glm 64 4 4 +``` + +Selected experts are uploaded during startup, so capacity failures occur before +inference. The budget is clamped against free VRAM after reserving the projected +dense resident set and 2 GB of runtime headroom per device. With `COLI_GPUS`, +`CUDA_EXPERT_GB` is a total budget across the device set; experts are assigned +whole to the least-loaded device that can hold them. Multi-GPU runs default to +`PIN_FILL=1` (measured hot set first, then unused VRAM filled with zero-heat +experts) and `CUDA_RELEASE_HOST=1` (RAM copy released after upload, reloaded +from disk only if CUDA later fails). + +`CUDA_EXPERT_GB=auto` fills each device up to measured free memory minus +projected dense tensors and headroom. `PIN_GB=all` then loads the remaining +routed experts into RAM **up to the `--ram` budget** (it clamps — [#229](https://github.com/JustVugg/colibri/issues/229)), +eliminating decode-time disk misses when capacity permits. This mode is intended +for dedicated high-memory inference hosts. + +### Full-residency reference result (6× RTX 5090, 251 GiB host) + +`CUDA_EXPERT_GB=auto PIN_GB=all` selected a 176.7 GB VRAM tier + 191.3 GB RAM +tier (all 19,456 experts resident), adapting the VRAM tier every 16 tokens. +With the GPU-resident pipeline (`COLI_CUDA_PIPE=2`) and Tensor-Core W4A16 +dispatch (`COLI_CUDA_TC_W4A16=1`), 96-token greedy decode measured +**5.8–6.8 tok/s** (TTFT ~13 s; 1571-token prefill ~122 s then 4.2 tok/s). +Full experiment log: [experiments/glm52-6x5090-2026-07-12.md](experiments/glm52-6x5090-2026-07-12.md). +These are host-specific capacity results, not portable defaults. + +## The GPU-resident pipeline (`COLI_CUDA_PIPE`) + +`COLI_CUDA_PIPE=2` keeps the residual stream on-device across layers: rmsnorms, +residual adds, router GEMMs and the shared expert run on the GPU while the CPU +expert loop runs uninterrupted, with batched attention and grouped expert +uploads at prefill. On a single-GPU host this also pays at decode (S=1): +**+49%** measured on a 5070 Ti ([#273](https://github.com/JustVugg/colibri/issues/273)/#274); +on multi-GPU hosts the per-layer P2P hops cancel the gain, so the decode gate is +device-count aware. `COLI_CUDA_TC_W4A16=1` enables Tensor-Core int4×fp16 mixed +dispatch for batched rows (pays at ≥16 rows). + +## Notes and limitations + +- Text-mode timing reports prefill separately from decode. +- MTP speculation defaults off on CUDA (cold draft routes increase expert + traffic); explicit `DRAFT=n` overrides. Since #294, `SPEC_PIN=1` keeps + draft/verify kernels consistent when speculation is on. +- Devices use independent contexts; a single expert is not sharded. Kernels are + correctness-first custom kernels. +- Profile quality matters more than raw VRAM capacity: the same 150 GB tier + measured 0.94–1.64 tok/s hot-first vs 0.29 tok/s filled without routing heat. +- The GPU tier earns its VRAM only when the CPU is the weak link — a tuned + AVX-512 CPU can match a 5090 on expert matmul + ([#101](https://github.com/JustVugg/colibri/issues/101)). + +## Reproducible backend A/B without the full checkpoint + +```bash +cd c +python tools/make_glm_bench_model.py --output /nvme/colibri-bench-medium --device cuda +python tools/benchmark_cuda_fixture.py --model /nvme/colibri-bench-medium --gpu 0 +``` + +The 313M-parameter fixture has random weights and is not a language model. It +preserves the real MLA/MoE/streaming shapes to compare CPU streaming, dense-only +CUDA, CPU hot-store, and CUDA hot-expert execution with identical replay tokens. diff --git a/docs/media/colibri-atlas.png b/docs/media/colibri-atlas.png new file mode 100644 index 0000000..5e5e76e Binary files /dev/null and b/docs/media/colibri-atlas.png differ diff --git a/docs/media/colibri-brain.png b/docs/media/colibri-brain.png index 0cd6c90..9ac5676 100644 Binary files a/docs/media/colibri-brain.png and b/docs/media/colibri-brain.png differ diff --git a/docs/media/colibri-dashboard.png b/docs/media/colibri-dashboard.png index 123fd3e..d51aa08 100644 Binary files a/docs/media/colibri-dashboard.png and b/docs/media/colibri-dashboard.png differ diff --git a/docs/media/colibri-metrics.png b/docs/media/colibri-metrics.png new file mode 100644 index 0000000..853d686 Binary files /dev/null and b/docs/media/colibri-metrics.png differ diff --git a/docs/media/colibri-mobile.png b/docs/media/colibri-mobile.png new file mode 100644 index 0000000..5c1c63c Binary files /dev/null and b/docs/media/colibri-mobile.png differ diff --git a/docs/media/ladder.png b/docs/media/ladder.png new file mode 100644 index 0000000..cd272fc Binary files /dev/null and b/docs/media/ladder.png differ diff --git a/docs/media/sparse.png b/docs/media/sparse.png new file mode 100644 index 0000000..0d7a4a1 Binary files /dev/null and b/docs/media/sparse.png differ diff --git a/docs/media/tiers.png b/docs/media/tiers.png new file mode 100644 index 0000000..9452587 Binary files /dev/null and b/docs/media/tiers.png differ diff --git a/docs/media/token-path.png b/docs/media/token-path.png new file mode 100644 index 0000000..4aebd39 Binary files /dev/null and b/docs/media/token-path.png differ diff --git a/docs/metal.md b/docs/metal.md new file mode 100644 index 0000000..3c94e5c --- /dev/null +++ b/docs/metal.md @@ -0,0 +1,30 @@ +# Metal backend (Apple Silicon, experimental) + +On Apple Silicon the decode profile is matmul-bound, and unified memory removes +the PCIe copy tax that keeps CUDA's streaming experts on the CPU — so colibrì +has an opt-in Metal backend that runs the **routed-expert SwiGLU (batched, +zero-copy from the RAM slabs)**, the **fused decode attention** (full MLA layer +in one command buffer, S≤4), and **prefill's large GEMMs** on the GPU. +Token-exact vs the CPU path. + +```bash +cd c +make glm METAL=1 # macOS only; no Xcode needed (shader compiles at runtime) +make metal-test # standalone kernel/attention correctness vs CPU reference +COLI_METAL=1 COLI_MODEL=/path/glm52_i4 ./coli chat --ram 96 +``` + +Measured on an M4 Max (128 GB, warm cache, MTP on): CPU 0.30 → Metal +**0.42 tok/s (~1.4×)** (best config adds `DIRECT=1`; ~3× vs this machine's +first cold run). An M5 Max with a 46.9 GB learned pin reached **2.06 tok/s** +([#103](https://github.com/JustVugg/colibri/issues/103); see also the +[M5 Max performance report](METAL-M5MAX-PERF-REPORT.md)). + +Key design points: Metal's ~5 ms submit latency makes per-matmul dispatch a +loss — everything is batched into few command buffers per layer, and the +resident experts' GPU work is submitted *before* the missed experts' disk reads +so I/O and compute overlap. `COLI_METAL_GEMM_MIN` tunes the prefill GEMM row +threshold (default 16). Streaming, cache, MTP, DSA and the persistence formats +are unchanged; every GPU path falls back to the CPU per-block on any fault. +Numerics are dequant→f32-MAC (same as the CUDA tier); greedy outputs are +byte-identical to the CPU engine. diff --git a/docs/tuning.md b/docs/tuning.md new file mode 100644 index 0000000..c95eca8 --- /dev/null +++ b/docs/tuning.md @@ -0,0 +1,110 @@ +# Tuning & runtime knobs + +Everything here is opt-in; the defaults are chosen so a plain `./coli chat` +is safe on any machine. See also [SETTINGS.md](SETTINGS.md) and +[ENVIRONMENT.md](ENVIRONMENT.md) for the full variable inventory. + +## The knobs that matter most + +| knob | what it does | +|---|---| +| `--temp T` | token sampling temperature (default 0.7 + nucleus 0.90 — tuned for int4; 0 = greedy) | +| `--topp 0.7` | adaptive expert top-p (30–40% less disk; lossy — prints a warning) | +| `--ngen N` | max tokens per answer (`:more` in chat continues a truncated one) | +| `--repin N` | adapt RAM/VRAM hot experts every N emitted tokens | +| `RAM_GB=` | claim more RAM for the expert cache than the conservative auto-detect | +| `PIN=stats PIN_GB=g` | pin the hottest experts from a measured usage profile | +| `DRAFT=n` | MTP draft depth (0 disables speculation) | +| `GRAMMAR=g.gbnf` | grammar-forced drafts for constrained JSON/NDJSON output ([docs](grammar-draft.md)) | +| `THINK=1` | enable GLM-5.2's reasoning block | +| `PILOT=1` | router-lookahead disk prefetch (see below) | +| `URING=1` | Linux-only batched expert I/O (implies `PIPE=1`) | +| `PIPE=0` | disable the async expert-load pool (default ON — overlaps `pread` with matmul, −18% disk service) | +| `DIRECT=1` | O_DIRECT expert reads (measured **+65%** alone on a Strix Halo, [#200](https://github.com/JustVugg/colibri/issues/200)) | +| `COLI_NUMA=1` | interleave resident weights across NUMA nodes on multi-socket hosts ([#82](https://github.com/JustVugg/colibri/issues/82)) | +| `CACHE_ROUTE=1` | cache-aware max-rank routing (opt-in, [#199](https://github.com/JustVugg/colibri/issues/199)) | +| `AUTOPIN=0` | disable the learning cache's auto-pin | +| `CAP_RAISE=0` | don't auto-grow the expert cache | +| `KVSAVE=0` | disable KV-cache persistence | +| `TF=1` | teacher-forcing validation | + +## Resource policy + +`coli plan` reports the planned hot (VRAM), warm (RAM), and cold backing (disk) +tiers, the reason for each placement, and the expected bottleneck. The default +`--policy quality` and `--policy balanced` modes preserve checkpoint quantization +and router decisions unless `--topk` or `--topp` is passed; those explicit lossy +overrides print a warning and proceed. + +Auto-tier plans size OpenMP from physical cores and bind workers across cores. +Memory-bound quantized kernels can regress sharply when SMT siblings compete for +limited memory channels; explicit `OMP_*` settings always take precedence. + +```bash +coli plan --model /models/glm52_i4 --policy quality +coli run --auto-tier --policy quality "Explain MoE offloading" +# Explicit research-only router reduction: +coli run --policy experimental-fast --topk 4 "Benchmark prompt" +``` + +Disk is an immutable recovery source, not a normal decode target. If the plan +leaves cold expert bytes on disk, speed depends on cache hit rate; output quality +does not. + +Cold expert reads can use a deferred pipeline: resident RAM/VRAM experts execute +while missing experts are loaded in a bounded background I/O pool, then the cold +results join before the layer completes. The pool engages only under `PIPE=1`; +`PIPE_WORKERS=n` sets its worker count (default 8). Profiling reports both disk +service time and the smaller foreground-visible wait time so overlap is explicit. + +`--policy balanced` enables lossless live placement (`REPIN=64`). At safe request +boundaries, a per-layer LFRU score combines decaying session frequency with recent +access and replaces at most four sufficiently colder pinned experts. `--policy +quality` leaves live replacement off by default; `REPIN=0` always disables it. + +## The learning cache + +The engine records which experts your usage actually routes to (`.coli_usage` +next to the model, updated every turn) and at startup automatically pins the +hottest ones in spare RAM — colibrì literally gets faster the more you use it. +`PIN=auto` seeds the pin directly from the live usage history +([#301](https://github.com/JustVugg/colibri/pull/301)). + +**The expert cache auto-sizes to your RAM** (since 2026-07-10): the engine +*raises* the LRU cap to fill your `--ram` budget instead of only lowering it. +If you benchmarked colibrì before that date, rerun — your numbers were capped. + +**Live tier adaptation** (`--repin N`, opt-in): at safe turn boundaries, a +decaying session heat map replaces cold pinned experts with hotter streamed +experts. A 25% hysteresis and a four-swap limit prevent tier thrashing. +Persistent `.coli_usage` remains the long-term signal and is not decayed. + +## Router-lookahead prefetch (`PILOT=1`, experimental) + +GLM-5.2's expert routing is measurably predictable *ahead of time* — applying +layer L+1's router to layer L's post-attention state recalls **71.6%** of the +true top-8 (vs 41.3% for "same experts as last token"). `PILOT=1` issues +next-layer expert readahead from a dedicated I/O thread while the current layer +computes. `PILOT_REAL=1` moves the prefetched loads off the critical path +(measured +11pp hit rate on a big-cache host), and `PILOT_TWO=1` folds the +computed shared-expert into the prediction (+3% recall, +[#200](https://github.com/JustVugg/colibri/issues/200)). On disk-saturated +hosts hint-only PILOT can be net negative — measure on yours. + +## Speculation and reproducibility + +Speculative decoding requires that the draft and verify paths compute the same +function — `SPEC_PIN=1` (default since [#294](https://github.com/JustVugg/colibri/pull/294)) +pins every forward issued while drafts are live to the platform's S=1 kernel +family. For byte-exact reproducibility across runs: `DRAFT=0`, plus `IDOT=0 +COLI_CUDA=0` if you also want kernel-family/GPU independence. Acceptance +percentages are not comparable across engine versions under `--topp` +([#163](https://github.com/JustVugg/colibri/issues/163) has the full story). + +## Conversations reopen warm + +`coli chat` persists the compressed MLA KV-cache to disk after every turn +(`.coli_kv`, ~182 KB/token, appended incrementally, crash-safe). Close the chat, +reopen it tomorrow — the model still remembers the whole conversation and **zero +re-prefill happens**: validated byte-identical to an uninterrupted session. +`:reset` clears it, `KVSAVE=0` disables it. diff --git a/docs/windows.md b/docs/windows.md new file mode 100644 index 0000000..94a4f69 --- /dev/null +++ b/docs/windows.md @@ -0,0 +1,90 @@ +# Windows 11 (native, no WSL) + +colibrì builds and runs natively on Windows 11 x86-64 with MinGW-w64. The port +adds a `_WIN32` compatibility layer in `c/compat.h` that maps POSIX I/O to the +Windows API (pread → ReadFile+OVERLAPPED, posix_fadvise no-op, aligned +allocation, MoveFileEx rename, GlobalMemoryStatusEx RAM detection). All platform +differences stay in `compat.h`; the engine source is unchanged. + +**Toolchain:** GCC via [winlibs](https://winlibs.com/) or MSYS2 MinGW-w64. +Tested with GCC 16.1.0 (x86_64-ucrt-posix-seh). + +```powershell +# One-time toolchain install (pick one): +scoop install mingw-winlibs # portable, no shell needed +# or: pacman -S mingw-w64-x86_64-gcc make # via MSYS2 + +# Build (from c/ directory): +make glm.exe # GLM-5.2 engine (static, no DLL dependencies) +make olmoe.exe # OLMoE engine (same shims) +make iobench.exe # disk I/O benchmark +make test-c # run C tests +make test-python # run Python tests (requires python) + +# AVX-VNNI: Intel Alder Lake+ (and Meteor Lake+) CPUs have a 128-bit int8 +# dot-product instruction (VPDPBUSD) the engine can use for ~1.3x faster +# quantized matmul. The x86-64-v3 default (portable AVX2) compiles it out; +# build for THIS machine to enable it: +make glm.exe ARCH=native # banner prints "idot: avx-vnni" + +# Verify (tiny model, 2.4 MB): +pip install torch transformers safetensors huggingface_hub +python tools/make_glm_oracle.py # generate tiny oracle +SNAP=./glm_tiny TF=1 ./glm.exe 64 16 16 # expect "32/32 positions" + +# Run with real model: +SNAP=D:\glm52_i4 ./glm.exe 64 4 16 # batch inference +python coli chat --model D:\glm52_i4 # interactive chat +python coli serve --model D:\glm52_i4 # OpenAI-compatible API +``` + +> Windows Store's `python` alias stub is the single most common native-Windows +> trap: install real Python (python.org or `winget install Python.Python.3.12`) +> or disable the alias under *Settings → Apps → App execution aliases*. + +## Warmup (overnight cache priming) + +The engine's expert cache learns from your workload. The included `warmup.ps1` +script runs `coli run` in a loop with diverse prompts to build the +`.coli_usage` histogram unattended, so the next real session starts with a +large, accurate hot-expert pin. Each run saves usage atomically on clean +completion. + +```powershell +.\warmup.ps1 -Rounds 1 -Ngen 32 # ~60-90 min, durable progress +``` + +## NVIDIA GPU (optional, via runtime DLL) + +On Windows the engine is built with MinGW gcc but CUDA kernels require MSVC + +nvcc. The split is clean: build the CUDA backend into a standalone +`coli_cuda.dll` (nvcc + MSVC), then the host `glm.exe` loads it at runtime via +`LoadLibrary` (`c/backend_loader.c`). The host never links cudart directly; if +the DLL is absent the engine falls back to CPU without error. + +```powershell +# Prerequisites: CUDA Toolkit + MSVC Build Tools (cl.exe) + nvcc on PATH. +# Build the DLL from a shell with the MSVC environment set (vcvars64.bat or +# "x64 Native Tools Command Prompt for VS"): +make cuda-dll CUDA_HOME="C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v12.8" CUDA_ARCH=sm_120 + +# Build the host with the runtime loader (CUDA_DLL=1 adds -DCOLI_CUDA and +# links backend_loader.o instead of cudart): +make glm.exe CUDA_DLL=1 ARCH=native + +# Run with the GPU expert tier (8 GB VRAM budget here; scale to your free VRAM): +$env:COLI_CUDA="1"; $env:COLI_GPU="0"; $env:CUDA_EXPERT_GB="8" +python coli chat --model D:\glm52_i4 --topp 0.7 +``` + +The DLL exports the full `extern "C"` surface (including the #111 pipeline ABI); +`backend_loader.c` resolves symbols via `GetProcAddress` on first use. +`ColiCudaTensor*` is opaque to the host (stored, never dereferenced), so the +MSVC-allocated struct is safe across the ABI boundary. `CUDA_ARCH` must match +your GPU's compute capability (e.g. `sm_120` for Blackwell / RTX 50-series, +`sm_89` for Ada / RTX 40-series). A one-shot `build_cuda.bat` wrapper is also +available. + +**Measured on a single RTX 5070 Ti + Core Ultra 9 (32 GB RAM):** CPU-only 0.63 +→ CUDA attention+dense 0.72 → **1.07 tok/s** with the GPU-resident pipeline at +decode ([#273](https://github.com/JustVugg/colibri/issues/273), merged in #274). diff --git a/flake.lock b/flake.lock new file mode 100644 index 0000000..9b788fa --- /dev/null +++ b/flake.lock @@ -0,0 +1,61 @@ +{ + "nodes": { + "flake-utils": { + "inputs": { + "systems": "systems" + }, + "locked": { + "lastModified": 1731533236, + "narHash": "sha256-l0KFg5HjrsfsO/JpG+r7fRrqm12kzFHyUHqHCVpMMbI=", + "owner": "numtide", + "repo": "flake-utils", + "rev": "11707dc2f618dd54ca8739b309ec4fc024de578b", + "type": "github" + }, + "original": { + "owner": "numtide", + "repo": "flake-utils", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1784160687, + "narHash": "sha256-iYL/bixrb6FlHFu/gIuBYzq6c6lM5AAXsXNSWXtIgQc=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "4382ed2b7a6839d4280a9b386db49cbc5907414d", + "type": "github" + }, + "original": { + "owner": "NixOS", + "ref": "nixos-26.05", + "repo": "nixpkgs", + "type": "github" + } + }, + "root": { + "inputs": { + "flake-utils": "flake-utils", + "nixpkgs": "nixpkgs" + } + }, + "systems": { + "locked": { + "lastModified": 1681028828, + "narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=", + "owner": "nix-systems", + "repo": "default", + "rev": "da67096a3b9bf56a91d16901293e51ba5b49a27e", + "type": "github" + }, + "original": { + "owner": "nix-systems", + "repo": "default", + "type": "github" + } + } + }, + "root": "root", + "version": 7 +} diff --git a/flake.nix b/flake.nix index 368b0a7..584a088 100644 --- a/flake.nix +++ b/flake.nix @@ -26,7 +26,9 @@ version = "1.0"; src = ./.; - nativeBuildInputs = [ pkgs.makeWrapper ]; + # python3 is needed by checkPhase: `make test-c` shells out to + # `python3 tools/run_tests.py` (see c/Makefile, PYTHON ?= python3). + nativeBuildInputs = [ pkgs.makeWrapper pkgs.python3 ]; buildInputs = [ pkgs.gcc @@ -44,18 +46,29 @@ installPhase = '' runHook preInstall - mkdir -p $out/bin - cp c/glm $out/bin/glm - # Wrap coli (the Python CLI) so it finds the right python and the engine - mkdir -p $out/share/colibri - cp c/coli $out/share/colibri/coli - chmod +x $out/share/colibri/coli - cp -r c/tools $out/share/colibri/tools + # Self-contained layout under $out/lib/colibri that mirrors the + # source tree `coli` runs in (see the path-resolution logic at the + # top of c/coli): the engine, the coli CLI script, the support + # modules it imports (openai_server.py, resource_plan.py, + # doctor.py), and tools/ all sit next to each other. + mkdir -p $out/lib/colibri/tools $out/bin + cp c/glm $out/lib/colibri/glm + cp c/coli $out/lib/colibri/coli + chmod +x $out/lib/colibri/coli + cp c/openai_server.py c/resource_plan.py c/doctor.py $out/lib/colibri/ + cp -r c/tools/* $out/lib/colibri/tools/ + # $out/bin holds the user-facing entry points. + ln -s ../lib/colibri/glm $out/bin/glm + + # Wrap coli: point it at the bundled engine (COLI_ENGINE) so it is + # found by default, and at the module dir (PYTHONPATH) so + # `import openai_server` / `resource_plan` / `doctor` resolve. makeWrapper ${pythonEnv}/bin/python $out/bin/coli \ - --add-flags "$out/share/colibri/coli" \ - --set PYTHONPATH "${pythonEnv}/${pkgs.python3.sitePackages}" + --add-flags "$out/lib/colibri/coli" \ + --set-default COLI_ENGINE "$out/lib/colibri/glm" \ + --set PYTHONPATH "$out/lib/colibri:${pythonEnv}/${pkgs.python3.sitePackages}" runHook postInstall '';