From d2d3a7b5598f0feb49ccc8e4665c2b7375d4afb2 Mon Sep 17 00:00:00 2001 From: Stonki13 <204933532+Stonki13@users.noreply.github.com> Date: Sat, 18 Jul 2026 23:18:16 +0200 Subject: [PATCH] Fix CUDA detection on Windows: cuda_binary()/cuda_linkage() always returned False MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both coli's cuda_binary() and doctor.py's cuda_linkage() detect CUDA support by running `ldd` on the engine binary and looking for a linked libcudart — but that's Linux-only (cuda_binary() checks sys.platform != "linux", and cuda_linkage() checks os.name != "posix", both short-circuiting to False on win32). Windows CUDA_DLL=1 builds never link libcudart at all: glm.exe loads coli_cuda.dll dynamically via LoadLibrary at startup (backend_loader.c), so there's no import-table entry for ldd/dumpbin to find in the first place. The practical effect: `coli doctor` always reported "NVIDIA GPU detected but the engine is CPU-only" on Windows, and `coli run/chat/serve --gpu ...` always hard-exited with "--gpu needs the CUDA build" — even on a correctly built CUDA_DLL=1 binary with coli_cuda.dll sitting right next to glm.exe. Fix: on win32, detect a COLI_CUDA build by scanning glm.exe for the marker string "[CUDA] mode: routed experts", which only exists in code compiled under #ifdef COLI_CUDA (see glm.c's cuda init block), then confirm coli_cuda.dll actually sits next to the binary — mirroring the Linux "linked but missing" distinction. Linux/macOS detection is unchanged. Verified on Windows 11 with mingw-w64 GCC 16.1.0 + CUDA 13.2 + RTX 5080: `coli doctor` now reports "CUDA engine and devices are available", and `coli run --gpu 0 --vram 8` populates VRAM (confirmed via the engine's own "[CUDA] resident set: N tensors, X GB VRAM" runtime log) instead of exiting. --- c/coli | 24 ++++++++++++++++++------ c/doctor.py | 35 ++++++++++++++++++++++++++--------- 2 files changed, 44 insertions(+), 15 deletions(-) diff --git a/c/coli b/c/coli index 4c27415..4566ea6 100755 --- a/c/coli +++ b/c/coli @@ -141,12 +141,24 @@ def need_model(model): sys.exit(f"{C.yel}engine is not built.{C.r} Run: coli build") def cuda_binary(): - if not os.path.exists(GLM) or sys.platform != "linux": return False - try: - linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3) - return any("libcudart" in line and "not found" not in line - for line in linked.stdout.splitlines()) - except (OSError,subprocess.SubprocessError): return False + if not os.path.exists(GLM): return False + if sys.platform == "linux": + try: + linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3) + return any("libcudart" in line and "not found" not in line + for line in linked.stdout.splitlines()) + except (OSError,subprocess.SubprocessError): return False + if sys.platform == "win32": + # Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads + # coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no + # import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a + # marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require + # coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup). + try: + with open(GLM,"rb") as f: built=b"[CUDA] mode: routed experts" in f.read() + except OSError: return False + return built and os.path.exists(os.path.join(os.path.dirname(GLM),"coli_cuda.dll")) + return False def resource_request(a, env): ctx=a.ctx or int(env.get("CTX",4096)) diff --git a/c/doctor.py b/c/doctor.py index 4cb5a42..cbe9c87 100644 --- a/c/doctor.py +++ b/c/doctor.py @@ -19,16 +19,33 @@ def _check(identifier, status, summary, **details): def cuda_linkage(engine_path): """Return CUDA linkage state without loading the executable or CUDA runtime.""" - if not Path(engine_path).is_file() or os.name != "posix": + engine = Path(engine_path) + if not engine.is_file(): return {"linked": False, "missing": False} - try: - result = subprocess.run(["ldd", str(engine_path)], capture_output=True, text=True, - timeout=3, check=False) - except (OSError, subprocess.SubprocessError): - return {"linked": False, "missing": False} - lines = [line for line in result.stdout.splitlines() if "libcudart" in line] - return {"linked": any("not found" not in line for line in lines), - "missing": any("not found" in line for line in lines)} + if os.name == "posix": + try: + result = subprocess.run(["ldd", str(engine)], capture_output=True, text=True, + timeout=3, check=False) + except (OSError, subprocess.SubprocessError): + return {"linked": False, "missing": False} + lines = [line for line in result.stdout.splitlines() if "libcudart" in line] + return {"linked": any("not found" not in line for line in lines), + "missing": any("not found" in line for line in lines)} + if sys.platform == "win32": + # Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads + # coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no + # import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a + # marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require + # coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup). + try: + built = b"[CUDA] mode: routed experts" in engine.read_bytes() + except OSError: + return {"linked": False, "missing": False} + if not built: + return {"linked": False, "missing": False} + dll_present = (engine.parent / "coli_cuda.dll").is_file() + return {"linked": dll_present, "missing": not dll_present} + return {"linked": False, "missing": False} def run_doctor(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0, *,