Spaces:
Running on Zero
Running on Zero
Download scripts/benchmark_kv_cache.py from Tejas123we/Saturday-AI: direct link, hf CLI and curl.
- Browser
- Download file 990 Bytes
-
https://huggingface.co/spaces/Tejas123we/Saturday-AI/resolve/main/scripts/benchmark_kv_cache.py
- Command line
-
hf download hf://spaces/Tejas123we/Saturday-AI/scripts/benchmark_kv_cache.py
-
curl -L -o benchmark_kv_cache.py https://huggingface.co/spaces/Tejas123we/Saturday-AI/resolve/main/scripts/benchmark_kv_cache.py
990 Bytes
| import time | |
| import numpy as np | |
| import argparse | |
| def benchmark(seq_len: int, new_tokens: int, use_cache: bool): | |
| print(f"Benchmarking (use_cache={use_cache}) ...") | |
| start = time.time() | |
| # Mock generation time | |
| # Without cache, time is O(n^2), with cache it's O(n) | |
| for i in range(new_tokens): | |
| if use_cache: | |
| time.sleep(0.001) | |
| else: | |
| time.sleep(0.001 + 0.0001 * (seq_len + i)) | |
| elapsed = time.time() - start | |
| tok_sec = new_tokens / elapsed | |
| print(f"Completed {new_tokens} tokens in {elapsed:.3f}s ({tok_sec:.2f} tok/s)\n") | |
| if __name__ == "__main__": | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--seq-len", type=int, default=128) | |
| parser.add_argument("--new-tokens", type=int, default=256) | |
| args = parser.parse_args() | |
| print("=== KV Cache Benchmark ===") | |
| benchmark(args.seq_len, args.new_tokens, use_cache=False) | |
| benchmark(args.seq_len, args.new_tokens, use_cache=True) | |