Fix CUDA detection on Windows: cuda_binary()/cuda_linkage() always returned False

Both coli's cuda_binary() and doctor.py's cuda_linkage() detect CUDA support
by running `ldd` on the engine binary and looking for a linked libcudart —
but that's Linux-only (cuda_binary() checks sys.platform != "linux", and
cuda_linkage() checks os.name != "posix", both short-circuiting to False on
win32). Windows CUDA_DLL=1 builds never link libcudart at all: glm.exe loads
coli_cuda.dll dynamically via LoadLibrary at startup (backend_loader.c), so
there's no import-table entry for ldd/dumpbin to find in the first place.

The practical effect: `coli doctor` always reported "NVIDIA GPU detected but
the engine is CPU-only" on Windows, and `coli run/chat/serve --gpu ...`
always hard-exited with "--gpu needs the CUDA build" — even on a correctly
built CUDA_DLL=1 binary with coli_cuda.dll sitting right next to glm.exe.

Fix: on win32, detect a COLI_CUDA build by scanning glm.exe for the marker
string "[CUDA] mode: routed experts", which only exists in code compiled
under #ifdef COLI_CUDA (see glm.c's cuda init block), then confirm
coli_cuda.dll actually sits next to the binary — mirroring the Linux
"linked but missing" distinction. Linux/macOS detection is unchanged.

Verified on Windows 11 with mingw-w64 GCC 16.1.0 + CUDA 13.2 + RTX 5080:
`coli doctor` now reports "CUDA engine and devices are available", and
`coli run --gpu 0 --vram 8` populates VRAM (confirmed via the engine's own
"[CUDA] resident set: N tensors, X GB VRAM" runtime log) instead of exiting.
This commit is contained in:
Stonki13
2026-07-18 23:18:16 +02:00
parent 72d3d37231
commit d2d3a7b559
2 changed files with 44 additions and 15 deletions
+18 -6
View File
@@ -141,12 +141,24 @@ def need_model(model):
sys.exit(f"{C.yel}engine is not built.{C.r} Run: coli build")
def cuda_binary():
if not os.path.exists(GLM) or sys.platform != "linux": return False
try:
linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3)
return any("libcudart" in line and "not found" not in line
for line in linked.stdout.splitlines())
except (OSError,subprocess.SubprocessError): return False
if not os.path.exists(GLM): return False
if sys.platform == "linux":
try:
linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3)
return any("libcudart" in line and "not found" not in line
for line in linked.stdout.splitlines())
except (OSError,subprocess.SubprocessError): return False
if sys.platform == "win32":
# Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads
# coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no
# import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a
# marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require
# coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup).
try:
with open(GLM,"rb") as f: built=b"[CUDA] mode: routed experts" in f.read()
except OSError: return False
return built and os.path.exists(os.path.join(os.path.dirname(GLM),"coli_cuda.dll"))
return False
def resource_request(a, env):
ctx=a.ctx or int(env.get("CTX",4096))
+26 -9
View File
@@ -19,16 +19,33 @@ def _check(identifier, status, summary, **details):
def cuda_linkage(engine_path):
"""Return CUDA linkage state without loading the executable or CUDA runtime."""
if not Path(engine_path).is_file() or os.name != "posix":
engine = Path(engine_path)
if not engine.is_file():
return {"linked": False, "missing": False}
try:
result = subprocess.run(["ldd", str(engine_path)], capture_output=True, text=True,
timeout=3, check=False)
except (OSError, subprocess.SubprocessError):
return {"linked": False, "missing": False}
lines = [line for line in result.stdout.splitlines() if "libcudart" in line]
return {"linked": any("not found" not in line for line in lines),
"missing": any("not found" in line for line in lines)}
if os.name == "posix":
try:
result = subprocess.run(["ldd", str(engine)], capture_output=True, text=True,
timeout=3, check=False)
except (OSError, subprocess.SubprocessError):
return {"linked": False, "missing": False}
lines = [line for line in result.stdout.splitlines() if "libcudart" in line]
return {"linked": any("not found" not in line for line in lines),
"missing": any("not found" in line for line in lines)}
if sys.platform == "win32":
# Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads
# coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no
# import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a
# marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require
# coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup).
try:
built = b"[CUDA] mode: routed experts" in engine.read_bytes()
except OSError:
return {"linked": False, "missing": False}
if not built:
return {"linked": False, "missing": False}
dll_present = (engine.parent / "coli_cuda.dll").is_file()
return {"linked": dll_present, "missing": not dll_present}
return {"linked": False, "missing": False}
def run_doctor(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0, *,