Download tools/drop_model_cache.py from HelloSun/sddqwen35a3b: direct link, hf CLI and curl.
- Browser
- Download file 1.61 kB
-
https://huggingface.co/HelloSun/sddqwen35a3b/resolve/main/tools/drop_model_cache.py
- Command line
-
hf download hf://HelloSun/sddqwen35a3b/tools/drop_model_cache.py
-
curl -L -o drop_model_cache.py https://huggingface.co/HelloSun/sddqwen35a3b/resolve/main/tools/drop_model_cache.py
1.61 kB
| #!/usr/bin/env python3 | |
| """丟掉模型檔的 kernel page cache(不需 root),讓 SSD 基準測量可重現。 | |
| 為什麼需要:Linux 會把讀過的檔案留在 page cache,下一次 benchmark 就會「命中核心的 | |
| RAM」而不是我們的 expert cache,測出來的 tok/s 就不是真的 SSD 速度。 | |
| 我們的分頁器本身在每次 pread 之後也會對該範圍丟 DONTNEED(SDQ_FADV_DONTNEED=1), | |
| 這個工具則是用來清掉「上一次殘留」的部分。 | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import ctypes | |
| import os | |
| from pathlib import Path | |
| POSIX_FADV_DONTNEED = 4 | |
| libc = ctypes.CDLL("libc.so.6", use_errno=True) | |
| def main() -> None: | |
| ap = argparse.ArgumentParser(description="丟掉某個檔案的 page cache") | |
| ap.add_argument("path", help="要清掉的檔案(通常是 .gguf)") | |
| args = ap.parse_args() | |
| p = Path(args.path) | |
| fd = os.open(p, os.O_RDONLY) | |
| try: | |
| # 先整個檔案丟一次 | |
| rc = libc.posix_fadvise(fd, 0, 0, POSIX_FADV_DONTNEED) | |
| # 再逐 256 MB 丟一次(某些核心對大範圍的處理不一致) | |
| chunk = 256 << 20 | |
| size = p.stat().st_size | |
| off = 0 | |
| while off < size: | |
| libc.posix_fadvise(fd, off, min(chunk, size - off), POSIX_FADV_DONTNEED) | |
| off += chunk | |
| rc2 = libc.posix_fadvise(fd, 0, 0, POSIX_FADV_DONTNEED) | |
| print(f"已對 {p} 發出 POSIX_FADV_DONTNEED(rc={rc}, rc2={rc2})," | |
| f"{size / 2**30:.2f} GiB 的 page cache 應已釋放") | |
| finally: | |
| os.close(fd) | |
| if __name__ == "__main__": | |
| main() |