From 0049ea15ae59fd17eefe7e75f6d9f9b407fbc385 Mon Sep 17 00:00:00 2001 From: monotophic Date: Sat, 18 Jul 2026 13:56:06 -0400 Subject: [PATCH] E5: MTLResidencySet lifecycle scaffolding (init/register/unregister/shutdown) Env-gated (COLI_METAL_RESSET=1) persistent MTLResidencySet attached to g_queue (macOS 15+, @available-guarded with a one-line stderr fallback). Adds resset_add/resset_remove/resset_flush helpers, wires them into the existing coli_metal_register/coli_metal_unregister/coli_metal_shutdown bodies -- no new functions in backend_metal.h, no glm.c changes, every existing call site keeps its current signature and behavior. register() defers the commit (resset_add just marks dirty, under the same g_slab_mtx that already serializes parallel OMP loader threads) so a loader burst doesn't pay a commit per slab; unregister() commits synchronously and immediately, because the caller frees the underlying host memory right after it returns and a deferred removal would leave the set referencing freed memory. Nothing yet reads g_resset_enabled to change dispatch behavior -- this commit is bookkeeping only, gate on or off, so it does not change what any command buffer does (verified by inspection: no useResource:/useHeap: call site is touched here). Gate off (default) is byte-for-byte the stock path: g_resset_enabled starts false and nothing sets it outside the COLI_METAL_RESSET branch in coli_metal_init, so every new helper is a no-op. See SUMMARY.md (next commit) for the full design and UNCERTAINTIES. --- c/backend_metal.mm | 81 ++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 78 insertions(+), 3 deletions(-) diff --git a/c/backend_metal.mm b/c/backend_metal.mm index 51ed8a0..b57fa66 100644 --- a/c/backend_metal.mm +++ b/c/backend_metal.mm @@ -258,6 +258,51 @@ extern "C" void coli_metal_attn_lat(double *ksched, double *gsched){ struct Slab { void *base; size_t len; id buf; }; static std::vector g_slabs; static std::mutex g_slab_mtx; // expert_load registers slabs from parallel OpenMP threads + +// ---- E5 experiment: COLI_METAL_RESSET=1 -- one persistent MTLResidencySet attached to +// g_queue (macOS 15+) replaces moe_submit's per-command-buffer useResource: loop over +// resolved expert weight/scale slabs. Allocation is untouched (same newBufferWithBytesNoCopy +// wrap as stock); only residency bookkeeping moves off the dispatch hot path -- see +// SUMMARY.md for why skipping useResource: there is safe (read-only, indirectly-referenced +// buffers only; residency sets don't do hazard tracking, but nothing here relied on it). +// g_resset_obj is a bare `id` (holds id) so the global's declared type +// carries no availability annotation -- the protocol name only appears inside +// @available(macOS 15.0, *) guards below, keeping -Wunguarded-availability clean. +static id g_resset_obj; +static bool g_resset_enabled; // COLI_METAL_RESSET=1, macOS 15+, and creation succeeded +static bool g_resset_dirty; // addAllocation: calls pending commit; g_slab_mtx-guarded + +// Add a just-wrapped buffer to the set. Deferred commit (batches an OMP loader burst into +// one commit at the next moe_submit instead of one per slab): correct because every +// register/unregister caller holds g_slab_mtx here, and resolve() (which every dispatch +// calls before touching a slab) takes the same mutex -- a slab moe_submit's resolve() can +// see was necessarily added, and marked dirty, before moe_submit's own resset_flush() ran. +static void resset_add(id b) { // caller holds g_slab_mtx + if (!g_resset_enabled) return; + if (@available(macOS 15.0, *)) { [(id)g_resset_obj addAllocation:b]; g_resset_dirty = true; } +} +// Remove + commit immediately, NOT deferred: the caller frees the underlying host memory +// right after coli_metal_unregister returns, so the removal must be applied before that -- +// an uncommitted-but-still-resident allocation pointing at freed memory is a use-after-free +// risk the GPU could act on. +static void resset_remove(id b) { // caller holds g_slab_mtx + if (!g_resset_enabled) return; + if (@available(macOS 15.0, *)) { + id rs = (id)g_resset_obj; + [rs removeAllocation:b]; [rs commit]; + } + g_resset_dirty = false; // commit above also flushes any pending adds +} +// Flush pending adds before moe_submit relies on the set alone for residency -- the only +// caller that skips per-buffer useResource: (see moe_submit below). +static void resset_flush() { + if (!g_resset_enabled) return; + std::lock_guard lk(g_slab_mtx); + if (!g_resset_dirty) return; + if (@available(macOS 15.0, *)) { [(id)g_resset_obj commit]; } + g_resset_dirty = false; +} + // Persistent scratch buffers (grow-only) for the MoE pipeline. static id g_gg, g_uu, g_hh, g_xg; static size_t g_gg_cap, g_uu_cap, g_hh_cap, g_xg_cap; static id ensure(id b, size_t *cap, size_t need) { @@ -304,6 +349,25 @@ extern "C" int coli_metal_init(void) { if (!g_gemv || !g_moe_gemv || !g_moe_silu || !g_a_rms || !g_a_rope || !g_a_copy || !g_a_qabs || !g_a_score || !g_a_smax || !g_a_clat || !g_a_ctx) { fprintf(stderr, "[metal] pipeline failed\n"); g_dev = nil; return 0; } + // E5 experiment: COLI_METAL_RESSET=1 -- see g_resset_obj comment above. + if (getenv("COLI_METAL_RESSET") && atoi(getenv("COLI_METAL_RESSET"))) { + if (@available(macOS 15.0, *)) { + MTLResidencySetDescriptor *rd = [MTLResidencySetDescriptor new]; + rd.initialCapacity = 4096; // hint only (internal array presize), not a hard limit + NSError *rerr = nil; + id rs = [g_dev newResidencySetWithDescriptor:rd error:&rerr]; + if (rs) { + [g_queue addResidencySet:rs]; + g_resset_obj = rs; g_resset_enabled = true; + fprintf(stderr, "[METAL] residency-set: on (macOS 15+, moe_submit skips per-buffer useResource:)\n"); + } else { + fprintf(stderr, "[METAL] residency-set create failed: %s -- stock per-CB residency path\n", + rerr ? [[rerr localizedDescription] UTF8String] : "?"); + } + } else { + fprintf(stderr, "[METAL] COLI_METAL_RESSET=1 requested but OS < macOS 15 -- stock per-CB residency path\n"); + } + } } return 1; } @@ -314,12 +378,16 @@ extern "C" void coli_metal_register(void *base, size_t len) { options:g_res_opts deallocator:nil]; if (!b) return; std::lock_guard lk(g_slab_mtx); // called from parallel expert_load threads - for (auto &s : g_slabs) if (s.base == base) { s.len = len; s.buf = b; return; } + for (auto &s : g_slabs) if (s.base == base) { s.len = len; s.buf = b; resset_add(b); return; } g_slabs.push_back({base, len, b}); + resset_add(b); } extern "C" void coli_metal_unregister(void *base) { std::lock_guard lk(g_slab_mtx); - for (size_t i=0;i resolve(const void *p, uint64_t *addr) { @@ -357,7 +425,14 @@ extern "C" void coli_metal_spin_start(void) { } extern "C" void coli_metal_spin_stop(void) { g_spin_run.store(false); } -extern "C" void coli_metal_shutdown(void) { coli_metal_spin_stop(); g_gemv=nil; g_queue=nil; g_dev=nil; g_tensor_count=g_tensor_bytes=0; } +extern "C" void coli_metal_shutdown(void) { + coli_metal_spin_stop(); + if (g_resset_enabled) { + if (@available(macOS 15.0, *)) { [g_queue removeResidencySet:(id)g_resset_obj]; } + } + g_resset_obj=nil; g_resset_enabled=false; g_resset_dirty=false; + g_gemv=nil; g_queue=nil; g_dev=nil; g_tensor_count=g_tensor_bytes=0; +} extern "C" int coli_metal_available(void) { return g_dev != nil; } extern "C" void coli_metal_stats(size_t *c, size_t *b) { if(c)*c=g_tensor_count; if(b)*b=g_tensor_bytes; } extern "C" int coli_metal_mem_info(size_t *used, size_t *total) {