| |
| """Run the calibration corpus through the live 1M model with the |
| VLLM_ACT_CAPTURE_DIR hook: per layer, ~24k routed tokens of (x, topk). |
| Reuses the corpus builder from collect_expert_stats_v2.""" |
| import os |
| import sys |
|
|
| os.environ["VLLM_ACT_CAPTURE_DIR"] = "/data/glm52-acts" |
| os.environ.setdefault("VLLM_PP_LAYER_PARTITION", "21,19,19,19") |
| sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) |
|
|
| from collect_expert_stats_v2 import build_corpus |
|
|
|
|
| def main(): |
| from vllm import LLM, SamplingParams |
| from vllm.inputs import TokensPrompt |
|
|
| llm = LLM( |
| model="/data/glm52", |
| pipeline_parallel_size=4, |
| gpu_memory_utilization=0.509, |
| kv_cache_dtype="fp8_ds_mla", |
| max_model_len=8192, |
| max_num_seqs=2, |
| max_num_batched_tokens=2048, |
| enforce_eager=True, |
| ) |
| tok = llm.get_tokenizer() |
| texts = build_corpus(tok) |
| prompts = [TokensPrompt(prompt_token_ids=tok.encode(t)[:7900]) |
| for t in texts] |
| print(f"{len(prompts)} calibration prompts") |
| llm.generate(prompts, SamplingParams(max_tokens=1, temperature=0.0)) |
| print("generation done; check /data/glm52-acts") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|