christopher-kapic's picture
Upload folder using huggingface_hub
fdc6474 verified
Raw
History Blame Contribute Delete
1.23 kB
#!/usr/bin/env python3
"""Run the calibration corpus through the live 1M model with the
VLLM_ACT_CAPTURE_DIR hook: per layer, ~24k routed tokens of (x, topk).
Reuses the corpus builder from collect_expert_stats_v2."""
import os
import sys
os.environ["VLLM_ACT_CAPTURE_DIR"] = "/data/glm52-acts"
os.environ.setdefault("VLLM_PP_LAYER_PARTITION", "21,19,19,19")
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from collect_expert_stats_v2 import build_corpus # noqa: E402
def main():
from vllm import LLM, SamplingParams
from vllm.inputs import TokensPrompt
llm = LLM(
model="/data/glm52",
pipeline_parallel_size=4,
gpu_memory_utilization=0.509,
kv_cache_dtype="fp8_ds_mla",
max_model_len=8192,
max_num_seqs=2,
max_num_batched_tokens=2048,
enforce_eager=True,
)
tok = llm.get_tokenizer()
texts = build_corpus(tok)
prompts = [TokensPrompt(prompt_token_ids=tok.encode(t)[:7900])
for t in texts]
print(f"{len(prompts)} calibration prompts")
llm.generate(prompts, SamplingParams(max_tokens=1, temperature=0.0))
print("generation done; check /data/glm52-acts")
if __name__ == "__main__":
main()