From cf062010d8dff7a7dd89174170ed33f6be92aa90 Mon Sep 17 00:00:00 2001 From: Robert Landers Date: Wed, 15 Jul 2026 04:07:12 +0200 Subject: [PATCH] use async and fix workers Signed-off-by: Robert Landers --- .gitignore | 1 + c/glm.c | 11 ++++++++--- c/tests/test_uring.c | 3 +++ c/uring.h | 11 +++++++++++ docs/ENVIRONMENT.md | 4 ++-- 5 files changed, 25 insertions(+), 5 deletions(-) diff --git a/.gitignore b/.gitignore index dba6a3f..981e740 100644 --- a/.gitignore +++ b/.gitignore @@ -67,3 +67,4 @@ c/bench/ c/tests/test_decode_batch c/tests/test_i4_acc512 c/tests/test_idot +c/tests/test_uring diff --git a/c/glm.c b/c/glm.c index 0237869..7d307dc 100644 --- a/c/glm.c +++ b/c/glm.c @@ -5259,6 +5259,7 @@ int main(int argc, char **argv){ g_pipe = getenv("PIPE")?atoi(getenv("PIPE")):0; /* default OFF: overlap expert load ‖ matmul (byte-identical; reorders I/O). PIPE=1 opts in */ g_pipe_nw = getenv("PIPE_WORKERS")?atoi(getenv("PIPE_WORKERS")):8; /* I/O worker threads */ if(g_pipe_nw<1) g_pipe_nw=1; + g_direct = getenv("DIRECT")?atoi(getenv("DIRECT")):0; g_uring = getenv("URING")?atoi(getenv("URING")):0; if(g_uring){ #ifdef __linux__ @@ -5267,13 +5268,17 @@ int main(int argc, char **argv){ if(uring_batch_init(&g_ub_pipe) || (g_pilot_real&&uring_batch_init(&g_ub_pilot))){ fprintf(stderr,"URING=1: io_uring_setup failed: %s\n",strerror(errno)); return 2; } - fprintf(stderr,"[URING] queued expert I/O active (depth=%d%s)\n",URING_REQ_MAX, - g_pilot_real?", batched PILOT_REAL":""); + unsigned uw=(unsigned)(g_pipe_nw>64?64:g_pipe_nw); + if(coli_uring_set_workers(&g_ub_pipe.ring,uw) || + (g_pilot_real&&coli_uring_set_workers(&g_ub_pilot.ring,uw))) + fprintf(stderr,"[URING] warning: cannot set io-wq workers=%u: %s\n",uw,strerror(errno)); + fprintf(stderr,"[URING] queued expert I/O active (depth=%d, workers=%u, %s%s)\n",URING_REQ_MAX,uw, + g_direct?"O_DIRECT":"buffered",g_pilot_real?", batched PILOT_REAL":""); + if(!g_direct) fprintf(stderr,"[URING] cold NVMe: DIRECT=1 avoids page-cache copy/readahead bottlenecks\n"); #else fprintf(stderr,"URING=1 is supported only on Linux\n"); return 2; #endif } - g_direct = getenv("DIRECT")?atoi(getenv("DIRECT")):0; g_idot = getenv("IDOT")?atoi(getenv("IDOT")):1; /* 0 = kernel f32 esatti (A/B) */ if(getenv("ROUTE_TRACE")&&*getenv("ROUTE_TRACE")){ g_route_fp=fopen(getenv("ROUTE_TRACE"),"w"); diff --git a/c/tests/test_uring.c b/c/tests/test_uring.c index 9712f65..e64b9f2 100644 --- a/c/tests/test_uring.c +++ b/c/tests/test_uring.c @@ -84,6 +84,9 @@ int main(void){ } close(fd); return fail("io_uring_setup"); } + if(coli_uring_set_workers(&ring,4)){ + coli_uring_close(&ring); close(fd); return fail("io-wq worker limit"); + } for(int i=0;ifd,IORING_REGISTER_IOWQ_MAX_WORKERS,limits,2); +} + static inline int coli_uring_prep_read(ColiUring *r,int fd,void *buf,size_t len, int64_t off,uint64_t user_data){ if(!len || len>UINT32_MAX){ errno=EINVAL; return -1; } @@ -91,6 +96,12 @@ static inline int coli_uring_prep_read(ColiUring *r,int fd,void *buf,size_t len, struct io_uring_sqe *sqe=&r->sqes[idx]; memset(sqe,0,sizeof(*sqe)); sqe->opcode=IORING_OP_READ; + /* Cold regular-file reads are allowed to execute inline during + * io_uring_enter() unless forced async. That serializes the submitter on + * filesystems without native nonblocking buffered reads and destroys the + * intended I/O/compute overlap. io-wq gives the ring a real bounded worker + * pool while CQEs retain completion ordering/ownership here. */ + sqe->flags=IOSQE_ASYNC; sqe->fd=fd; sqe->off=(uint64_t)off; sqe->addr=(uint64_t)(uintptr_t)buf; diff --git a/docs/ENVIRONMENT.md b/docs/ENVIRONMENT.md index d280a92..dab656b 100644 --- a/docs/ENVIRONMENT.md +++ b/docs/ENVIRONMENT.md @@ -39,8 +39,8 @@ Format: `VAR` — default — effect. | `COLI_METAL_GEMM_MIN` | `16` | Minimum matmul rows to dispatch a GEMM to the GPU (below this, stays on CPU). | | `COLI_METAL_SPIN` | off | Keep a GPU keep-alive spinner running (reduces dispatch latency; costs power). | | `PIPE` | `0` (off) | Overlap expert disk-load with matmul via I/O worker threads. Byte-identical output; reorders I/O. `PIPE=1` opts in. | -| `PIPE_WORKERS` | `8` | Number of I/O worker threads when `PIPE=1`. Tune to your SSD (fewer avoids over-subscribing cores). | -| `URING` | `0` (off) | Linux-only queued expert I/O. `URING=1` implies `PIPE=1`, replaces blocking loader pthreads and spin waits with batched io_uring SQEs/CQEs, and batches `PILOT_REAL` loads on a separate ring. Fails clearly if the kernel denies io_uring; incompatible with `COLI_MMAP=1`. | +| `PIPE_WORKERS` | `8` | Number of pthread loaders when `PIPE=1`, or the io-wq worker maximum per ring when `URING=1` (capped at 64). Tune to SSD queue depth and available cores. | +| `URING` | `0` (off) | Linux-only queued expert I/O. `URING=1` implies `PIPE=1`, forces cold reads through io-wq (`IOSQE_ASYNC`), replaces blocking loader pthreads and spin waits with batched SQEs/CQEs, and batches `PILOT_REAL` loads on a separate ring. Use `DIRECT=1` for cold NVMe to avoid page-cache copy/readahead limits. Fails clearly if the kernel denies io_uring; incompatible with `COLI_MMAP=1`. | | `DIRECT` | `0` (off) | Use `O_DIRECT`/unbuffered reads for expert slabs. Helps sustained NVMe; keeps the zero-copy GPU path. | | `COLI_NO_OMP_TUNE` | off | **Kill-switch** for the OpenMP hot-thread tuning (`OMP_WAIT_POLICY=active` spin + proc-bind). Set `=1` when the CPU is mostly waiting on the GPU (Metal) so spin doesn't steal the shared power budget. | | `MLOCK` | `-1` (auto: on for macOS) | Wire the streamed expert cache into physical RAM (`mlock`) to dodge the memory compressor. `0` off, `1` force. |