test1111111 / scripts /machine_report.py
spitfire4794's picture
CISM remote autobench: full source + fleet runner, serve results on 7860
28a1a01
Raw History Blame Contribute Delete
20.2 kB
#!/usr/bin/env python
"""Machine capability report for deciding where CISM would run best.
Stdlib-only by default; NumPy (if installed) enables measured bandwidth and
sync probes. Works on Windows, Linux, and macOS without admin rights. Run:
python scripts/machine_report.py # human-readable report
python scripts/machine_report.py --json # machine-readable
python scripts/machine_report.py --fast # smaller/faster probes
Interpretation guide is printed at the end: cache sizes set which model
footprints can be cache-resident, the bandwidth ladder shows the measured
cache-to-RAM multiplier on this machine, and the sync probe estimates the
per-barrier cost that penalizes multithreaded sub-millisecond decoding.
"""
from __future__ import annotations
import argparse
import json
import os
import platform
import statistics
import sys
import threading
import time
def collect_platform() -> dict:
info = {
"os": platform.system(),
"os_release": platform.release(),
"python": platform.python_version(),
"machine": platform.machine(),
}
if info["os"] == "Windows":
try:
import winreg
with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE,
r"HARDWARE\DESCRIPTION\System\CentralProcessor\0") as key:
info["cpu_model"] = winreg.QueryValueEx(key, "ProcessorNameString")[0]
except OSError:
info["cpu_model"] = platform.processor()
elif info["os"] == "Linux":
for line in _read_file("/proc/cpuinfo").splitlines():
if line.lower().startswith("model name"):
info["cpu_model"] = line.split(":", 1)[1].strip()
break
else:
info["cpu_model"] = _sysctl("machdep.cpu.brand_string") or platform.processor()
return info
def collect_cpu_topology() -> dict:
info: dict = {}
system = platform.system()
if system == "Windows":
try:
info["logical_processors"] = os.cpu_count()
result = _powershell(
"Get-CimInstance Win32_Processor | Select-Object -First 1 "
"NumberOfCores,NumberOfLogicalProcessors,L2CacheSize,L3CacheSize,MaxClockSpeed "
"| ConvertTo-Json -Compress")
import json as _json
data = _json.loads(result)
info["physical_cores"] = int(data["NumberOfCores"])
info["logical_processors"] = int(data["NumberOfLogicalProcessors"])
if data.get("L2CacheSize"):
info["l2_per_package_kb"] = int(data["L2CacheSize"])
if data.get("L3CacheSize"):
info["l3_total_kb"] = int(data["L3CacheSize"])
if data.get("MaxClockSpeed"):
info["max_clock_mhz"] = int(data["MaxClockSpeed"])
except Exception as error:
info["windows_topology_error"] = str(error)
elif system == "Linux":
info["logical_processors"] = os.cpu_count()
cpuinfo = _read_file("/proc/cpuinfo")
cores = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines()
if line.strip().startswith("core id")}
physical_ids = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines()
if line.strip().startswith("physical id")}
info["physical_cores"] = len(cores) * max(len(physical_ids), 1) if cores else None
info["max_clock_mhz"] = max((float(line.split(":", 1)[1]) for line in cpuinfo.splitlines()
if line.strip().startswith("cpu MHz")), default=None)
info["cache"] = _cache_linux()
else:
info["logical_processors"] = os.cpu_count()
info["cache"] = _cache_macos()
if system == "Windows" and "cache" not in info:
try:
info["cache"] = _cache_windows()
except Exception as error:
info["cache_error"] = str(error)
return info
def _read_file(path: str) -> str:
try:
with open(path, encoding="utf-8", errors="replace") as handle:
return handle.read()
except OSError:
return ""
def _sysctl(name: str) -> str | None:
try:
import subprocess
return subprocess.run(["sysctl", "-n", name], capture_output=True,
text=True, timeout=5).stdout.strip()
except Exception:
return None
def _powershell(command: str) -> str:
import subprocess
result = subprocess.run(
["powershell", "-NoProfile", "-NonInteractive", "-Command", command],
capture_output=True, text=True, timeout=30)
if result.returncode != 0:
raise RuntimeError(result.stderr.strip() or "powershell failed")
return result.stdout.strip()
def _cache_linux() -> dict:
caches: dict[int, int] = {}
line_sizes: dict[int, int] = {}
for index in range(16):
base = f"/sys/devices/system/cpu/cpu0/cache/index{index}"
text = _read_file(f"{base}/size").strip()
if not text:
continue
level = int(_read_file(f"{base}/level").strip())
size_kb = int(text.rstrip("KkMm")) * (1024 if text.lower().endswith("m") else 1)
cache_type = _read_file(f"{base}/type").strip()
if cache_type in ("Data", "Unified") or level >= 3:
caches[level] = max(caches.get(level, 0), size_kb)
try:
line_sizes[level] = int(_read_file(f"{base}/coherency_line_size").strip())
except (OSError, ValueError):
pass
return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())},
"line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())},
"shared_l3_note": "L3 value is per-CCX/package as exposed by sysfs"}
def _cache_windows() -> dict:
import ctypes
import struct
kernel32 = ctypes.windll.kernel32
length = ctypes.c_ulong(0)
kernel32.GetLogicalProcessorInformation(None, ctypes.byref(length))
if length.value == 0:
raise RuntimeError("GetLogicalProcessorInformation returned empty length")
buffer = (ctypes.c_char * length.value)()
if not kernel32.GetLogicalProcessorInformation(buffer, ctypes.byref(length)):
raise RuntimeError("GetLogicalProcessorInformation failed")
count = length.value // 24 if platform.machine().endswith("64") else length.value // 24
# x64 entry layout: mask(8) + relationship(4) + padding(4) + union(16,
# two ULONGLONGs) = 32 bytes; x86 uses 24. Never guess from the reported
# length - derive stride from the platform ABI.
stride = 32 if platform.machine().endswith("64") else 24
count = length.value // stride
caches: dict[int, int] = {}
line_sizes: dict[int, int] = {}
for entry in range(count):
offset = entry * stride
relationship = struct.unpack_from("<I", buffer, offset + 8)[0]
if relationship != 2: # RelationCache
continue
base = offset + 16
level = struct.unpack_from("<B", buffer, base)[0]
if not 1 <= level <= 3: # Reject misaligned garbage instead of keying it.
continue
line_size = struct.unpack_from("<H", buffer, base + 2)[0]
size_kb = struct.unpack_from("<I", buffer, base + 4)[0] // 1024
cache_type = struct.unpack_from("<i", buffer, base + 8)[0]
# Type 1 = instruction, 2 = data, 0 = unified.
if level >= 3 or cache_type in (0, 2):
caches[level] = max(caches.get(level, 0), size_kb)
line_sizes[level] = line_size
return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())},
"line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())}}
def _cache_macos() -> dict:
caches = {}
for level, name in ((1, "hw.l1icachesize"), (2, "hw.l2cachesize"), (3, "hw.l3cachesize")):
value = _sysctl(name)
if value and value.isdigit():
caches[f"l{level}"] = int(value) // 1024
return {"per_level_kb": caches}
def collect_isa() -> dict:
# Prefer CISM's native CPUID probe: precise and vendor-agnostic (reports
# VNNI/AVX-512 correctly on Windows, where the OS exposes almost nothing).
try:
import cism._native as native
features = dict(native.cpu_features())
if features:
features["neon"] = False
features["asimddp"] = False
return features
except Exception:
pass
system = platform.system()
flags: set[str] = set()
if system == "Linux":
for line in _read_file("/proc/cpuinfo").splitlines():
if line.lower().startswith("flags"):
flags = set(line.split(":", 1)[1].lower().split())
break
elif system == "Windows":
import ctypes
# PF_AVX2_INSTRUCTIONS_AVAILABLE = 40, PF_AVX512F_INSTRUCTIONS_AVAILABLE = 43.
present = ctypes.windll.kernel32.IsProcessorFeaturePresent
flags = {"avx2"} if present(40) else set()
if present(43):
flags.add("avx512f")
flags.add("fma") if flags else flags # AVX2 CPUs in practice ship FMA.
else:
features = _sysctl("machdep.cpu.features") or ""
flags = set(features.lower().split())
result = {name: (name in flags) for name in
("avx2", "fma", "avx512f", "avx512_vnni", "avx512_bf16")}
# ARM rows (e.g. Apple platforms running x86-unaware Pythons) via os flags.
result["neon"] = "neon" in flags or "asimd" in flags
result["asimddp"] = "asimddp" in flags or "dotprod" in flags
return result
def collect_memory() -> dict:
system = platform.system()
if system == "Windows":
import ctypes
class MemoryStatus(ctypes.Structure):
_fields_ = [("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong),
("ullTotalPhys", ctypes.c_ulonglong), ("ullAvailPhys", ctypes.c_ulonglong),
("ullTotalPageFile", ctypes.c_ulonglong), ("ullAvailPageFile", ctypes.c_ulonglong),
("ullTotalVirtual", ctypes.c_ulonglong), ("ullAvailVirtual", ctypes.c_ulonglong),
("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
status = MemoryStatus()
status.dwLength = ctypes.sizeof(MemoryStatus)
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(status))
return {"total_gb": round(status.ullTotalPhys / 1e9, 2),
"available_gb": round(status.ullAvailPhys / 1e9, 2)}
if system == "Linux":
total_kb, available_kb = 0, 0
for line in _read_file("/proc/meminfo").splitlines():
if line.startswith("MemTotal"):
total_kb = int(line.split()[1])
if line.startswith("MemAvailable"):
available_kb = int(line.split()[1])
return {"total_gb": round(total_kb / 1e6, 2), "available_gb": round(available_kb / 1e6, 2)}
value = _sysctl("hw.memsize")
return {"total_gb": round(int(value) / 1e9, 2)} if value and value.isdigit() else {}
def collect_virtualization() -> dict:
info: dict = {"detected": None, "vcpu_quota": None}
system = platform.system()
if system == "Linux":
cpuinfo = _read_file("/proc/cpuinfo").lower()
if "hypervisor" in cpuinfo or "vmware" in cpuinfo or "kvm" in cpuinfo:
info["detected"] = "hypervisor flag in /proc/cpuinfo"
cpu_max = _read_file("/sys/fs/cgroup/cpu.max").strip()
if cpu_max and not cpu_max.startswith("max"):
quota, period = cpu_max.split()
info["vcpu_quota"] = round(int(quota) / int(period), 2)
else:
quota = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_quota_us").strip()
period = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_period_us").strip()
if quota and quota != "-1" and period:
info["vcpu_quota"] = round(int(quota) / int(period), 2)
elif system == "Windows":
try:
result = _powershell(
"Get-CimInstance Win32_ComputerSystem | Select-Object -First 1 "
"Manufacturer,Model | ConvertTo-Json -Compress")
import json as _json
data = _json.loads(result)
identity = f"{data.get('Manufacturer', '')} {data.get('Model', '')}"
info["system_identity"] = identity.strip()
markers = ("virtual", "vmware", "kvm", "qemu", "xen", "hyper-v", "bochs")
if any(marker in identity.lower() for marker in markers):
info["detected"] = identity.strip()
except Exception:
pass
return info
def probe_bandwidth(size_mb: float, reps: int = 5) -> float:
"""Single-core sequential read bandwidth via numpy count_nonzero, in GB/s.
uint8 count_nonzero is a pure streaming read; float32 arithmetic reductions
are compute-limited and undercount real bandwidth about 5x.
"""
import numpy as np
buffer = np.zeros(int(size_mb * 1e6), dtype=np.uint8)
buffer[::4096] = 1 # Defeat trivially-skippable all-zero fast paths.
best = float("inf")
for _ in range(reps):
start = time.perf_counter()
found = np.count_nonzero(buffer)
elapsed = time.perf_counter() - start
best = min(best, elapsed)
if found == buffer.size:
raise RuntimeError("impossible count")
return buffer.nbytes / best / 1e9
def probe_thread_sync(rounds: int = 20000) -> float:
"""Ping-pong between two threads; returns microseconds per round trip."""
ping = threading.Event()
pong = threading.Event()
ping.set()
def responder():
for _ in range(rounds):
ping.wait()
ping.clear()
pong.set()
worker = threading.Thread(target=responder, daemon=True)
worker.start()
start = time.perf_counter()
for _ in range(rounds):
pong.wait()
pong.clear()
ping.set()
worker.join(timeout=5)
elapsed = time.perf_counter() - start
return elapsed / rounds * 1e6
def cism_probe() -> dict | None:
try:
import cism
import cism._native as native
except ImportError:
return None
info: dict = {"version": cism.__version__}
# Smoke check with a tiny in-memory model; not a performance claim.
config = {"model_type": "llama", "hidden_size": 24, "intermediate_size": 37,
"num_hidden_layers": 2, "num_attention_heads": 4, "num_key_value_heads": 2,
"head_dim": 8, "vocab_size": 43, "max_position_embeddings": 80,
"rms_norm_eps": 1e-5, "rope_theta": 10000.0, "tie_word_embeddings": True}
try:
import numpy as np
rng = np.random.default_rng(7)
weights = {"model.embed_tokens.weight": rng.normal(0, 0.1, (43, 24)).astype(np.float32),
"model.norm.weight": np.ones(24, dtype=np.float32)}
for layer in range(2):
prefix = f"model.layers.{layer}."
weights[prefix + "input_layernorm.weight"] = np.ones(24, dtype=np.float32)
weights[prefix + "post_attention_layernorm.weight"] = np.ones(24, dtype=np.float32)
weights[prefix + "self_attn.q_proj.weight"] = rng.normal(0, 0.1, (32, 24)).astype(np.float32)
weights[prefix + "self_attn.k_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32)
weights[prefix + "self_attn.v_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32)
weights[prefix + "self_attn.o_proj.weight"] = rng.normal(0, 0.1, (24, 32)).astype(np.float32)
weights[prefix + "mlp.gate_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32)
weights[prefix + "mlp.up_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32)
weights[prefix + "mlp.down_proj.weight"] = rng.normal(0, 0.1, (24, 37)).astype(np.float32)
model = native.Model(config, weights, "fp32")
session = model.create_session([1, 2, 3], max_new_tokens=2, temperature=0,
top_p=1, top_k=0, seed=0, eos_token_ids=[])
tokens = session.next_tokens(2)
info["native_smoke_test"] = "pass" if len(tokens) == 2 else "no tokens"
info["kernel"] = model.info["kernel"]
info["threads"] = model.info["threads"]
info["precision"] = model.info["precision"]
except Exception as error:
info["native_smoke_test"] = f"fail: {error}"
return info
def verdict(data: dict) -> list[str]:
lines = []
cache = (data.get("cpu") or {}).get("cache") or {}
per_level = cache.get("per_level_kb", {})
l3_kb = per_level.get("l3")
if l3_kb:
l3_mb = l3_kb / 1024
lines.append(f"L3 cache: {l3_mb:.1f} MB -> CISM cache-resident ceiling is about "
f"{l3_mb * 0.8:.0f}-{l3_mb * 0.9:.0f} MB of packed weights "
"(leaving headroom for KV, activations, and the OS).")
ladder = data.get("bandwidth_ladder", {})
rates = [value.get("gbps") for value in ladder.values() if value.get("gbps")]
if len(rates) >= 3:
small, large = rates[0], rates[-1]
if large > 0:
lines.append(f"Measured bandwidth ladder: {rates[0]:.1f} -> {rates[1]:.1f} -> "
f"{rates[2]:.1f} GB/s (small -> mid -> large buffer). "
f"Cache-to-RAM multiplier on this machine: about {small / large:.1f}x.")
sync = data.get("thread_sync_us_per_round")
if sync is not None:
lines.append(f"Thread ping-pong: {sync:.2f} us/round trip. Sub-microsecond models "
f"(<= 1 MB) will lose most of their time to barriers like this; "
f"single-thread them.")
isa = data.get("isa", {})
if isa.get("avx512_vnni"):
lines.append("AVX-512 VNNI available: integer-int8 kernels can use hardware "
"dot products here.")
elif isa.get("avx2"):
lines.append("AVX2 available: CISM uses the FMA path (no hardware VNNI; "
"expect compute-bound int8 kernels).")
quota = (data.get("virtualization") or {}).get("vcpu_quota")
if quota is not None:
lines.append(f"cgroup vCPU quota: {quota:.2f} CPUs - multithreaded decode beyond "
f"this quota will be throttled.")
return lines
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--json", action="store_true", help="emit JSON only")
parser.add_argument("--fast", action="store_true", help="smaller probe buffers")
parser.add_argument("--skip-bench", action="store_true", help="skip measured probes")
args = parser.parse_args()
report: dict = {"platform": collect_platform(), "cpu": collect_cpu_topology(),
"isa": collect_isa(), "memory": collect_memory(),
"virtualization": collect_virtualization()}
if not args.skip_bench:
try:
import numpy # noqa: F401
except ImportError:
report["bandwidth_ladder"] = {"error": "numpy is required for probes"}
else:
rungs = ((0.5, "l2_class"), (16, "l3_class"), (256, "dram_class")) if not args.fast \
else ((0.5, "l2_class"), (4, "l3_class"), (64, "dram_class"))
ladder = {}
for size_mb, name in rungs:
try:
ladder[name] = {"buffer_mb": size_mb,
"gbps": round(probe_bandwidth(size_mb), 2)}
except Exception as error:
ladder[name] = {"error": str(error)}
report["bandwidth_ladder"] = ladder
report["thread_sync_us_per_round"] = round(probe_thread_sync(), 3)
report["cism"] = cism_probe()
report["interpretation"] = verdict(report)
if args.json:
print(json.dumps(report, indent=2))
return 0
print(json.dumps(report, indent=2))
print("\n" + "=" * 78)
print("INTERPRETATION")
print("=" * 78)
for line in report["interpretation"]:
print(f"- {line}")
return 0
if __name__ == "__main__":
raise SystemExit(main())