Merge pull request #395 from Stonki13/fix/windows-cuda-detection

Fix CUDA detection on Windows (coli/doctor.py always report CPU-only)
This commit is contained in:
Vincenzo Fornaro
2026-07-19 10:40:01 +02:00
committed by GitHub
2 changed files with 44 additions and 15 deletions
+18 -6
View File
@@ -143,12 +143,24 @@ def need_model(model):
sys.exit(f"{C.yel}engine is not built.{C.r} Run: coli build")
def cuda_binary():
if not os.path.exists(GLM) or sys.platform != "linux": return False
try:
linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3)
return any("libcudart" in line and "not found" not in line
for line in linked.stdout.splitlines())
except (OSError,subprocess.SubprocessError): return False
if not os.path.exists(GLM): return False
if sys.platform == "linux":
try:
linked=subprocess.run(["ldd",GLM],capture_output=True,text=True,timeout=3)
return any("libcudart" in line and "not found" not in line
for line in linked.stdout.splitlines())
except (OSError,subprocess.SubprocessError): return False
if sys.platform == "win32":
# Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads
# coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no
# import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a
# marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require
# coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup).
try:
with open(GLM,"rb") as f: built=b"[CUDA] mode: routed experts" in f.read()
except OSError: return False
return built and os.path.exists(os.path.join(os.path.dirname(GLM),"coli_cuda.dll"))
return False
def resource_request(a, env):
ctx=a.ctx or int(env.get("CTX",4096))