Merge pull request #363 from woolcoxm/fix/win32-coli-chat-cuda-autoenable
win32: auto-enable the GPU in bare coli chat
This commit is contained in:
@@ -32,9 +32,16 @@ def args(**over):
|
||||
|
||||
|
||||
class EnvDefaultsTest(unittest.TestCase):
|
||||
def env_for_with(self, environ, platform):
|
||||
def env_for_with(self, environ, platform, cuda=False):
|
||||
"""Run env_for on a bare-chat args() under a faked env + platform.
|
||||
|
||||
cuda=False by default so the existing default-I/O tests stay
|
||||
deterministic: the Windows auto-enable branch calls cuda_binary() and
|
||||
(if True) discover_gpus(), both of which reach the real machine — faking
|
||||
False keeps these tests independent of the host's GPU."""
|
||||
with mock.patch.dict(os.environ, environ, clear=True), \
|
||||
mock.patch.object(sys, "platform", platform):
|
||||
mock.patch.object(sys, "platform", platform), \
|
||||
mock.patch.object(coli, "cuda_binary", return_value=cuda):
|
||||
return coli.env_for(args())
|
||||
|
||||
def test_win32_sets_measured_defaults(self):
|
||||
@@ -64,5 +71,80 @@ class EnvDefaultsTest(unittest.TestCase):
|
||||
self.assertNotIn(k, e)
|
||||
|
||||
|
||||
class CudaAutoEnableTest(unittest.TestCase):
|
||||
"""Windows bare `coli chat` (no --gpu/--vram/--auto-tier) used to ALWAYS run
|
||||
CPU-only even on a CUDA build with a GPU present. env_for now auto-enables
|
||||
CUDA on win32 when cuda_binary() is True and a GPU is discoverable; falls
|
||||
back to CPU with a warning if nvidia-smi (discover_gpus) is missing; stays
|
||||
silent on a CPU build; and never touches the Linux path."""
|
||||
|
||||
def _env_for(self, platform, cuda, gpus, plan=None):
|
||||
# Patch discover_gpus / build_plan / environment_for_plan at the
|
||||
# resource_plan module (env_for imports them lazily on each call, so the
|
||||
# patches are live when those imports run). Stubbing the planner keeps
|
||||
# the test independent of a real model dir (args().model == "X").
|
||||
import resource_plan
|
||||
a = args()
|
||||
GPB = 1024 ** 3
|
||||
if plan is None:
|
||||
plan = {"tiers": {"ram": {"budget_bytes": 16 * GPB, "cache_slots_per_layer": 4},
|
||||
"vram": {"budget_bytes": int(8.0 * GPB), "devices": gpus}}}
|
||||
|
||||
def fake_environment_for_plan(p, env, cuda_enabled=True):
|
||||
# Mirror the real contract: size CUDA_EXPERT_GB from the plan's VRAM
|
||||
# budget (this is the value env_for propagates into the engine env).
|
||||
r = dict(env)
|
||||
if cuda_enabled and p["tiers"]["vram"]["devices"] and p["tiers"]["vram"]["budget_bytes"] > 0:
|
||||
r["CUDA_EXPERT_GB"] = f"{p['tiers']['vram']['budget_bytes'] / GPB:.3f}"
|
||||
return r
|
||||
|
||||
with mock.patch.dict(os.environ, {}, clear=True), \
|
||||
mock.patch.object(sys, "platform", platform), \
|
||||
mock.patch.object(coli, "cuda_binary", return_value=cuda), \
|
||||
mock.patch.object(resource_plan, "discover_gpus", return_value=gpus), \
|
||||
mock.patch.object(resource_plan, "build_plan", return_value=plan), \
|
||||
mock.patch.object(resource_plan, "environment_for_plan",
|
||||
side_effect=fake_environment_for_plan):
|
||||
return coli.env_for(a)
|
||||
|
||||
def _fake_gpu(self, index=0, name="NVIDIA GeForce RTX 5070 Ti",
|
||||
total_mib=16384, free_mib=15000):
|
||||
return {"index": index, "name": name,
|
||||
"total_bytes": total_mib * 1024 * 1024,
|
||||
"free_bytes": free_mib * 1024 * 1024}
|
||||
|
||||
def test_win32_auto_enables_cuda_when_gpu_present(self):
|
||||
e = self._env_for("win32", cuda=True, gpus=[self._fake_gpu()])
|
||||
self.assertEqual(e["COLI_CUDA"], "1")
|
||||
self.assertEqual(e["COLI_GPUS"], "0")
|
||||
# VRAM budget is sized from free VRAM by build_plan (real minus reserve),
|
||||
# so it must be present and positive — never a guess or zero.
|
||||
self.assertIn("CUDA_EXPERT_GB", e)
|
||||
self.assertGreater(float(e["CUDA_EXPERT_GB"]), 0.0)
|
||||
# Dense offload is an explicit opt-in (matches --auto-tier): not set here.
|
||||
self.assertNotIn("CUDA_DENSE", e)
|
||||
|
||||
def test_win32_falls_back_to_cpu_when_nvidia_smi_missing(self):
|
||||
# coli_cuda.dll present (cuda=True) but nvidia-smi absent (no GPUs found)
|
||||
# -> warn + CPU-only, never crash, never set COLI_CUDA.
|
||||
e = self._env_for("win32", cuda=True, gpus=[])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("COLI_GPUS", e)
|
||||
self.assertNotIn("CUDA_EXPERT_GB", e)
|
||||
|
||||
def test_win32_cpu_build_stays_silent(self):
|
||||
# No coli_cuda.dll (cuda=False) -> CPU build, nothing GPU-related emitted.
|
||||
e = self._env_for("win32", cuda=False, gpus=[self._fake_gpu()])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("COLI_GPUS", e)
|
||||
|
||||
def test_linux_bare_chat_not_auto_enabled(self):
|
||||
# The auto-enable is scoped to win32: a Linux bare chat with a GPU
|
||||
# present must NOT turn CUDA on (Linux keeps the explicit-flag UX).
|
||||
e = self._env_for("linux", cuda=True, gpus=[self._fake_gpu()])
|
||||
self.assertNotIn("COLI_CUDA", e)
|
||||
self.assertNotIn("CUDA_EXPERT_GB", e)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user