#!/usr/bin/env python """Machine capability report for deciding where CISM would run best. Stdlib-only by default; NumPy (if installed) enables measured bandwidth and sync probes. Works on Windows, Linux, and macOS without admin rights. Run: python scripts/machine_report.py # human-readable report python scripts/machine_report.py --json # machine-readable python scripts/machine_report.py --fast # smaller/faster probes Interpretation guide is printed at the end: cache sizes set which model footprints can be cache-resident, the bandwidth ladder shows the measured cache-to-RAM multiplier on this machine, and the sync probe estimates the per-barrier cost that penalizes multithreaded sub-millisecond decoding. """ from __future__ import annotations import argparse import json import os import platform import statistics import sys import threading import time def collect_platform() -> dict: info = { "os": platform.system(), "os_release": platform.release(), "python": platform.python_version(), "machine": platform.machine(), } if info["os"] == "Windows": try: import winreg with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"HARDWARE\DESCRIPTION\System\CentralProcessor\0") as key: info["cpu_model"] = winreg.QueryValueEx(key, "ProcessorNameString")[0] except OSError: info["cpu_model"] = platform.processor() elif info["os"] == "Linux": for line in _read_file("/proc/cpuinfo").splitlines(): if line.lower().startswith("model name"): info["cpu_model"] = line.split(":", 1)[1].strip() break else: info["cpu_model"] = _sysctl("machdep.cpu.brand_string") or platform.processor() return info def collect_cpu_topology() -> dict: info: dict = {} system = platform.system() if system == "Windows": try: info["logical_processors"] = os.cpu_count() result = _powershell( "Get-CimInstance Win32_Processor | Select-Object -First 1 " "NumberOfCores,NumberOfLogicalProcessors,L2CacheSize,L3CacheSize,MaxClockSpeed " "| ConvertTo-Json -Compress") import json as _json data = _json.loads(result) info["physical_cores"] = int(data["NumberOfCores"]) info["logical_processors"] = int(data["NumberOfLogicalProcessors"]) if data.get("L2CacheSize"): info["l2_per_package_kb"] = int(data["L2CacheSize"]) if data.get("L3CacheSize"): info["l3_total_kb"] = int(data["L3CacheSize"]) if data.get("MaxClockSpeed"): info["max_clock_mhz"] = int(data["MaxClockSpeed"]) except Exception as error: info["windows_topology_error"] = str(error) elif system == "Linux": info["logical_processors"] = os.cpu_count() cpuinfo = _read_file("/proc/cpuinfo") cores = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines() if line.strip().startswith("core id")} physical_ids = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines() if line.strip().startswith("physical id")} info["physical_cores"] = len(cores) * max(len(physical_ids), 1) if cores else None info["max_clock_mhz"] = max((float(line.split(":", 1)[1]) for line in cpuinfo.splitlines() if line.strip().startswith("cpu MHz")), default=None) info["cache"] = _cache_linux() else: info["logical_processors"] = os.cpu_count() info["cache"] = _cache_macos() if system == "Windows" and "cache" not in info: try: info["cache"] = _cache_windows() except Exception as error: info["cache_error"] = str(error) return info def _read_file(path: str) -> str: try: with open(path, encoding="utf-8", errors="replace") as handle: return handle.read() except OSError: return "" def _sysctl(name: str) -> str | None: try: import subprocess return subprocess.run(["sysctl", "-n", name], capture_output=True, text=True, timeout=5).stdout.strip() except Exception: return None def _powershell(command: str) -> str: import subprocess result = subprocess.run( ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], capture_output=True, text=True, timeout=30) if result.returncode != 0: raise RuntimeError(result.stderr.strip() or "powershell failed") return result.stdout.strip() def _cache_linux() -> dict: caches: dict[int, int] = {} line_sizes: dict[int, int] = {} for index in range(16): base = f"/sys/devices/system/cpu/cpu0/cache/index{index}" text = _read_file(f"{base}/size").strip() if not text: continue level = int(_read_file(f"{base}/level").strip()) size_kb = int(text.rstrip("KkMm")) * (1024 if text.lower().endswith("m") else 1) cache_type = _read_file(f"{base}/type").strip() if cache_type in ("Data", "Unified") or level >= 3: caches[level] = max(caches.get(level, 0), size_kb) try: line_sizes[level] = int(_read_file(f"{base}/coherency_line_size").strip()) except (OSError, ValueError): pass return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())}, "line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())}, "shared_l3_note": "L3 value is per-CCX/package as exposed by sysfs"} def _cache_windows() -> dict: import ctypes import struct kernel32 = ctypes.windll.kernel32 length = ctypes.c_ulong(0) kernel32.GetLogicalProcessorInformation(None, ctypes.byref(length)) if length.value == 0: raise RuntimeError("GetLogicalProcessorInformation returned empty length") buffer = (ctypes.c_char * length.value)() if not kernel32.GetLogicalProcessorInformation(buffer, ctypes.byref(length)): raise RuntimeError("GetLogicalProcessorInformation failed") count = length.value // 24 if platform.machine().endswith("64") else length.value // 24 # x64 entry layout: mask(8) + relationship(4) + padding(4) + union(16, # two ULONGLONGs) = 32 bytes; x86 uses 24. Never guess from the reported # length - derive stride from the platform ABI. stride = 32 if platform.machine().endswith("64") else 24 count = length.value // stride caches: dict[int, int] = {} line_sizes: dict[int, int] = {} for entry in range(count): offset = entry * stride relationship = struct.unpack_from("= 3 or cache_type in (0, 2): caches[level] = max(caches.get(level, 0), size_kb) line_sizes[level] = line_size return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())}, "line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())}} def _cache_macos() -> dict: caches = {} for level, name in ((1, "hw.l1icachesize"), (2, "hw.l2cachesize"), (3, "hw.l3cachesize")): value = _sysctl(name) if value and value.isdigit(): caches[f"l{level}"] = int(value) // 1024 return {"per_level_kb": caches} def collect_isa() -> dict: # Prefer CISM's native CPUID probe: precise and vendor-agnostic (reports # VNNI/AVX-512 correctly on Windows, where the OS exposes almost nothing). try: import cism._native as native features = dict(native.cpu_features()) if features: features["neon"] = False features["asimddp"] = False return features except Exception: pass system = platform.system() flags: set[str] = set() if system == "Linux": for line in _read_file("/proc/cpuinfo").splitlines(): if line.lower().startswith("flags"): flags = set(line.split(":", 1)[1].lower().split()) break elif system == "Windows": import ctypes # PF_AVX2_INSTRUCTIONS_AVAILABLE = 40, PF_AVX512F_INSTRUCTIONS_AVAILABLE = 43. present = ctypes.windll.kernel32.IsProcessorFeaturePresent flags = {"avx2"} if present(40) else set() if present(43): flags.add("avx512f") flags.add("fma") if flags else flags # AVX2 CPUs in practice ship FMA. else: features = _sysctl("machdep.cpu.features") or "" flags = set(features.lower().split()) result = {name: (name in flags) for name in ("avx2", "fma", "avx512f", "avx512_vnni", "avx512_bf16")} # ARM rows (e.g. Apple platforms running x86-unaware Pythons) via os flags. result["neon"] = "neon" in flags or "asimd" in flags result["asimddp"] = "asimddp" in flags or "dotprod" in flags return result def collect_memory() -> dict: system = platform.system() if system == "Windows": import ctypes class MemoryStatus(ctypes.Structure): _fields_ = [("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong), ("ullTotalPhys", ctypes.c_ulonglong), ("ullAvailPhys", ctypes.c_ulonglong), ("ullTotalPageFile", ctypes.c_ulonglong), ("ullAvailPageFile", ctypes.c_ulonglong), ("ullTotalVirtual", ctypes.c_ulonglong), ("ullAvailVirtual", ctypes.c_ulonglong), ("ullAvailExtendedVirtual", ctypes.c_ulonglong)] status = MemoryStatus() status.dwLength = ctypes.sizeof(MemoryStatus) ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(status)) return {"total_gb": round(status.ullTotalPhys / 1e9, 2), "available_gb": round(status.ullAvailPhys / 1e9, 2)} if system == "Linux": total_kb, available_kb = 0, 0 for line in _read_file("/proc/meminfo").splitlines(): if line.startswith("MemTotal"): total_kb = int(line.split()[1]) if line.startswith("MemAvailable"): available_kb = int(line.split()[1]) return {"total_gb": round(total_kb / 1e6, 2), "available_gb": round(available_kb / 1e6, 2)} value = _sysctl("hw.memsize") return {"total_gb": round(int(value) / 1e9, 2)} if value and value.isdigit() else {} def collect_virtualization() -> dict: info: dict = {"detected": None, "vcpu_quota": None} system = platform.system() if system == "Linux": cpuinfo = _read_file("/proc/cpuinfo").lower() if "hypervisor" in cpuinfo or "vmware" in cpuinfo or "kvm" in cpuinfo: info["detected"] = "hypervisor flag in /proc/cpuinfo" cpu_max = _read_file("/sys/fs/cgroup/cpu.max").strip() if cpu_max and not cpu_max.startswith("max"): quota, period = cpu_max.split() info["vcpu_quota"] = round(int(quota) / int(period), 2) else: quota = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_quota_us").strip() period = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_period_us").strip() if quota and quota != "-1" and period: info["vcpu_quota"] = round(int(quota) / int(period), 2) elif system == "Windows": try: result = _powershell( "Get-CimInstance Win32_ComputerSystem | Select-Object -First 1 " "Manufacturer,Model | ConvertTo-Json -Compress") import json as _json data = _json.loads(result) identity = f"{data.get('Manufacturer', '')} {data.get('Model', '')}" info["system_identity"] = identity.strip() markers = ("virtual", "vmware", "kvm", "qemu", "xen", "hyper-v", "bochs") if any(marker in identity.lower() for marker in markers): info["detected"] = identity.strip() except Exception: pass return info def probe_bandwidth(size_mb: float, reps: int = 5) -> float: """Single-core sequential read bandwidth via numpy count_nonzero, in GB/s. uint8 count_nonzero is a pure streaming read; float32 arithmetic reductions are compute-limited and undercount real bandwidth about 5x. """ import numpy as np buffer = np.zeros(int(size_mb * 1e6), dtype=np.uint8) buffer[::4096] = 1 # Defeat trivially-skippable all-zero fast paths. best = float("inf") for _ in range(reps): start = time.perf_counter() found = np.count_nonzero(buffer) elapsed = time.perf_counter() - start best = min(best, elapsed) if found == buffer.size: raise RuntimeError("impossible count") return buffer.nbytes / best / 1e9 def probe_thread_sync(rounds: int = 20000) -> float: """Ping-pong between two threads; returns microseconds per round trip.""" ping = threading.Event() pong = threading.Event() ping.set() def responder(): for _ in range(rounds): ping.wait() ping.clear() pong.set() worker = threading.Thread(target=responder, daemon=True) worker.start() start = time.perf_counter() for _ in range(rounds): pong.wait() pong.clear() ping.set() worker.join(timeout=5) elapsed = time.perf_counter() - start return elapsed / rounds * 1e6 def cism_probe() -> dict | None: try: import cism import cism._native as native except ImportError: return None info: dict = {"version": cism.__version__} # Smoke check with a tiny in-memory model; not a performance claim. config = {"model_type": "llama", "hidden_size": 24, "intermediate_size": 37, "num_hidden_layers": 2, "num_attention_heads": 4, "num_key_value_heads": 2, "head_dim": 8, "vocab_size": 43, "max_position_embeddings": 80, "rms_norm_eps": 1e-5, "rope_theta": 10000.0, "tie_word_embeddings": True} try: import numpy as np rng = np.random.default_rng(7) weights = {"model.embed_tokens.weight": rng.normal(0, 0.1, (43, 24)).astype(np.float32), "model.norm.weight": np.ones(24, dtype=np.float32)} for layer in range(2): prefix = f"model.layers.{layer}." weights[prefix + "input_layernorm.weight"] = np.ones(24, dtype=np.float32) weights[prefix + "post_attention_layernorm.weight"] = np.ones(24, dtype=np.float32) weights[prefix + "self_attn.q_proj.weight"] = rng.normal(0, 0.1, (32, 24)).astype(np.float32) weights[prefix + "self_attn.k_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32) weights[prefix + "self_attn.v_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32) weights[prefix + "self_attn.o_proj.weight"] = rng.normal(0, 0.1, (24, 32)).astype(np.float32) weights[prefix + "mlp.gate_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32) weights[prefix + "mlp.up_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32) weights[prefix + "mlp.down_proj.weight"] = rng.normal(0, 0.1, (24, 37)).astype(np.float32) model = native.Model(config, weights, "fp32") session = model.create_session([1, 2, 3], max_new_tokens=2, temperature=0, top_p=1, top_k=0, seed=0, eos_token_ids=[]) tokens = session.next_tokens(2) info["native_smoke_test"] = "pass" if len(tokens) == 2 else "no tokens" info["kernel"] = model.info["kernel"] info["threads"] = model.info["threads"] info["precision"] = model.info["precision"] except Exception as error: info["native_smoke_test"] = f"fail: {error}" return info def verdict(data: dict) -> list[str]: lines = [] cache = (data.get("cpu") or {}).get("cache") or {} per_level = cache.get("per_level_kb", {}) l3_kb = per_level.get("l3") if l3_kb: l3_mb = l3_kb / 1024 lines.append(f"L3 cache: {l3_mb:.1f} MB -> CISM cache-resident ceiling is about " f"{l3_mb * 0.8:.0f}-{l3_mb * 0.9:.0f} MB of packed weights " "(leaving headroom for KV, activations, and the OS).") ladder = data.get("bandwidth_ladder", {}) rates = [value.get("gbps") for value in ladder.values() if value.get("gbps")] if len(rates) >= 3: small, large = rates[0], rates[-1] if large > 0: lines.append(f"Measured bandwidth ladder: {rates[0]:.1f} -> {rates[1]:.1f} -> " f"{rates[2]:.1f} GB/s (small -> mid -> large buffer). " f"Cache-to-RAM multiplier on this machine: about {small / large:.1f}x.") sync = data.get("thread_sync_us_per_round") if sync is not None: lines.append(f"Thread ping-pong: {sync:.2f} us/round trip. Sub-microsecond models " f"(<= 1 MB) will lose most of their time to barriers like this; " f"single-thread them.") isa = data.get("isa", {}) if isa.get("avx512_vnni"): lines.append("AVX-512 VNNI available: integer-int8 kernels can use hardware " "dot products here.") elif isa.get("avx2"): lines.append("AVX2 available: CISM uses the FMA path (no hardware VNNI; " "expect compute-bound int8 kernels).") quota = (data.get("virtualization") or {}).get("vcpu_quota") if quota is not None: lines.append(f"cgroup vCPU quota: {quota:.2f} CPUs - multithreaded decode beyond " f"this quota will be throttled.") return lines def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--json", action="store_true", help="emit JSON only") parser.add_argument("--fast", action="store_true", help="smaller probe buffers") parser.add_argument("--skip-bench", action="store_true", help="skip measured probes") args = parser.parse_args() report: dict = {"platform": collect_platform(), "cpu": collect_cpu_topology(), "isa": collect_isa(), "memory": collect_memory(), "virtualization": collect_virtualization()} if not args.skip_bench: try: import numpy # noqa: F401 except ImportError: report["bandwidth_ladder"] = {"error": "numpy is required for probes"} else: rungs = ((0.5, "l2_class"), (16, "l3_class"), (256, "dram_class")) if not args.fast \ else ((0.5, "l2_class"), (4, "l3_class"), (64, "dram_class")) ladder = {} for size_mb, name in rungs: try: ladder[name] = {"buffer_mb": size_mb, "gbps": round(probe_bandwidth(size_mb), 2)} except Exception as error: ladder[name] = {"error": str(error)} report["bandwidth_ladder"] = ladder report["thread_sync_us_per_round"] = round(probe_thread_sync(), 3) report["cism"] = cism_probe() report["interpretation"] = verdict(report) if args.json: print(json.dumps(report, indent=2)) return 0 print(json.dumps(report, indent=2)) print("\n" + "=" * 78) print("INTERPRETATION") print("=" * 78) for line in report["interpretation"]: print(f"- {line}") return 0 if __name__ == "__main__": raise SystemExit(main())