profile effective CPU expert bandwidth

This commit is contained in:
ZacharyZcR
2026-07-18 05:34:41 +08:00
parent c3a90eca36
commit 570d738ed5
3 changed files with 22 additions and 10 deletions
+2 -2
View File
@@ -6,7 +6,7 @@ from tools.benchmark_cuda_fixture import parse_output, parse_p0
SAMPLE = """
REPLAY decode: 4 tokens | 12.34 tok/s
PROFILE: expert-disk 1.25s | expert-matmul 2.50s | attention 0.75s | lm_head 0.10s | other -0.05s
P0-EXEC: routed CPU 1.200s | routed GPU critical 0.150s | router 0.200s | residual P2P 0.030s / 75 hop | orchestration 0.100s
P0-EXEC: routed CPU 1.200s / 123.40 GB/s (456 row) | routed GPU critical 0.150s | router 0.200s | residual P2P 0.030s / 75 hop | orchestration 0.100s
"""
@@ -21,7 +21,7 @@ class ParseOutputTest(unittest.TestCase):
parse_output("REPLAY decode: 4 tokens | 12.34 tok/s", "engine failed")
def test_extracts_p0_profile(self):
self.assertEqual(parse_p0(SAMPLE), [1.2, 0.15, 0.2, 0.03, 75.0, 0.1])
self.assertEqual(parse_p0(SAMPLE), [1.2, 123.4, 456.0, 0.15, 0.2, 0.03, 75.0, 0.1])
if __name__ == "__main__":