Merge pull request #363 from woolcoxm/fix/win32-coli-chat-cuda-autoenable

win32: auto-enable the GPU in bare coli chat
This commit is contained in:
Vincenzo Fornaro
2026-07-20 17:39:09 +02:00
committed by GitHub
3 changed files with 130 additions and 2 deletions
+84 -2
View File
@@ -32,9 +32,16 @@ def args(**over):
class EnvDefaultsTest(unittest.TestCase):
def env_for_with(self, environ, platform):
def env_for_with(self, environ, platform, cuda=False):
"""Run env_for on a bare-chat args() under a faked env + platform.
cuda=False by default so the existing default-I/O tests stay
deterministic: the Windows auto-enable branch calls cuda_binary() and
(if True) discover_gpus(), both of which reach the real machine — faking
False keeps these tests independent of the host's GPU."""
with mock.patch.dict(os.environ, environ, clear=True), \
mock.patch.object(sys, "platform", platform):
mock.patch.object(sys, "platform", platform), \
mock.patch.object(coli, "cuda_binary", return_value=cuda):
return coli.env_for(args())
def test_win32_sets_measured_defaults(self):
@@ -64,5 +71,80 @@ class EnvDefaultsTest(unittest.TestCase):
self.assertNotIn(k, e)
class CudaAutoEnableTest(unittest.TestCase):
"""Windows bare `coli chat` (no --gpu/--vram/--auto-tier) used to ALWAYS run
CPU-only even on a CUDA build with a GPU present. env_for now auto-enables
CUDA on win32 when cuda_binary() is True and a GPU is discoverable; falls
back to CPU with a warning if nvidia-smi (discover_gpus) is missing; stays
silent on a CPU build; and never touches the Linux path."""
def _env_for(self, platform, cuda, gpus, plan=None):
# Patch discover_gpus / build_plan / environment_for_plan at the
# resource_plan module (env_for imports them lazily on each call, so the
# patches are live when those imports run). Stubbing the planner keeps
# the test independent of a real model dir (args().model == "X").
import resource_plan
a = args()
GPB = 1024 ** 3
if plan is None:
plan = {"tiers": {"ram": {"budget_bytes": 16 * GPB, "cache_slots_per_layer": 4},
"vram": {"budget_bytes": int(8.0 * GPB), "devices": gpus}}}
def fake_environment_for_plan(p, env, cuda_enabled=True):
# Mirror the real contract: size CUDA_EXPERT_GB from the plan's VRAM
# budget (this is the value env_for propagates into the engine env).
r = dict(env)
if cuda_enabled and p["tiers"]["vram"]["devices"] and p["tiers"]["vram"]["budget_bytes"] > 0:
r["CUDA_EXPERT_GB"] = f"{p['tiers']['vram']['budget_bytes'] / GPB:.3f}"
return r
with mock.patch.dict(os.environ, {}, clear=True), \
mock.patch.object(sys, "platform", platform), \
mock.patch.object(coli, "cuda_binary", return_value=cuda), \
mock.patch.object(resource_plan, "discover_gpus", return_value=gpus), \
mock.patch.object(resource_plan, "build_plan", return_value=plan), \
mock.patch.object(resource_plan, "environment_for_plan",
side_effect=fake_environment_for_plan):
return coli.env_for(a)
def _fake_gpu(self, index=0, name="NVIDIA GeForce RTX 5070 Ti",
total_mib=16384, free_mib=15000):
return {"index": index, "name": name,
"total_bytes": total_mib * 1024 * 1024,
"free_bytes": free_mib * 1024 * 1024}
def test_win32_auto_enables_cuda_when_gpu_present(self):
e = self._env_for("win32", cuda=True, gpus=[self._fake_gpu()])
self.assertEqual(e["COLI_CUDA"], "1")
self.assertEqual(e["COLI_GPUS"], "0")
# VRAM budget is sized from free VRAM by build_plan (real minus reserve),
# so it must be present and positive — never a guess or zero.
self.assertIn("CUDA_EXPERT_GB", e)
self.assertGreater(float(e["CUDA_EXPERT_GB"]), 0.0)
# Dense offload is an explicit opt-in (matches --auto-tier): not set here.
self.assertNotIn("CUDA_DENSE", e)
def test_win32_falls_back_to_cpu_when_nvidia_smi_missing(self):
# coli_cuda.dll present (cuda=True) but nvidia-smi absent (no GPUs found)
# -> warn + CPU-only, never crash, never set COLI_CUDA.
e = self._env_for("win32", cuda=True, gpus=[])
self.assertNotIn("COLI_CUDA", e)
self.assertNotIn("COLI_GPUS", e)
self.assertNotIn("CUDA_EXPERT_GB", e)
def test_win32_cpu_build_stays_silent(self):
# No coli_cuda.dll (cuda=False) -> CPU build, nothing GPU-related emitted.
e = self._env_for("win32", cuda=False, gpus=[self._fake_gpu()])
self.assertNotIn("COLI_CUDA", e)
self.assertNotIn("COLI_GPUS", e)
def test_linux_bare_chat_not_auto_enabled(self):
# The auto-enable is scoped to win32: a Linux bare chat with a GPU
# present must NOT turn CUDA on (Linux keeps the explicit-flag UX).
e = self._env_for("linux", cuda=True, gpus=[self._fake_gpu()])
self.assertNotIn("COLI_CUDA", e)
self.assertNotIn("CUDA_EXPERT_GB", e)
if __name__ == "__main__":
unittest.main()